#!/usr/bin/env python3 """Deterministic medium-size fixtures, assembled only from DocView test content. These exercise page/chapter counts and local-resource navigation. They are not 1 GiB scan stress documents or a statistically representative book collection. """ from pathlib import Path import hashlib import json import zipfile from pypdf import PdfReader, PdfWriter root = Path(__file__).resolve().parent out = root / "fixtures/performance" out.mkdir(parents=True, exist_ok=True) pdf_root = root / "fixtures/pdf/extended" sources = [PdfReader(pdf_root / name) for name in ("japanese-layout.pdf", "graphics-compositing.pdf")] writer = PdfWriter() for index in range(500): source = sources[index % len(sources)] writer.add_page(source.pages[index % len(source.pages)]) writer.add_metadata({"/Title": "DocView generated 500-page performance corpus", "/Author": "DocView tests"}) with (out / "standard-500.pdf").open("wb") as handle: writer.write(handle) # A separate navigation document keeps the established page/graphics corpus # unchanged while providing 500 explicit, source-authored outline destinations. heading_writer = PdfWriter(clone_from=out / "standard-500.pdf") for index in range(500): heading_writer.add_outline_item(f"Generated chapter {index + 1}", index) with (out / "standard-headings-500.pdf").open("wb") as handle: heading_writer.write(handle) css = "body{font:18px sans-serif;line-height:1.6;margin:3em;max-width:55em}h1,h2{color:#165f69}img{width:240px}" svg = '' paragraph = "日本語の読み取りとローカル資源を確認します。This is original synthetic reading content." def html(chapter, sections=12): body = f'

第{chapter + 1}章

' body += ''.join(f'

節 {chapter + 1}.{i + 1}

{paragraph} ({chapter}/{i})

検証用の円
' for i in range(sections)) return f'章{chapter + 1}{body}' html_dir = out / "html" html_dir.mkdir(exist_ok=True) (html_dir / "index.html").write_text(html(0, 400), encoding="utf-8") (html_dir / "style.css").write_text(css, encoding="utf-8") (html_dir / "image.svg").write_text(svg, encoding="utf-8") def archive(name, entries): with zipfile.ZipFile(out / name, "w") as handle: for path, content in entries: info = zipfile.ZipInfo(path, (2026, 1, 1, 0, 0, 0)) info.create_system = 3 info.external_attr = 0o100600 << 16 info.compress_type = zipfile.ZIP_STORED if path == "mimetype" else zipfile.ZIP_DEFLATED handle.writestr(info, content.encode("utf-8")) archive("standard-html.zip", [("index.html", html(0, 400)), ("style.css", css), ("image.svg", svg)] + [(f"chapter-{i}.html", html(i)) for i in range(1, 50)]) manifest = ''.join(f'' for i in range(500)) spine = ''.join(f'' for i in range(500)) nav = ''.join(f'
  • 第{i + 1}章
  • ' for i in range(500)) package = f'urn:docview:generated:500生成した500章の資料ja{manifest}{spine}' epub_entries = [("mimetype", "application/epub+zip"), ("META-INF/container.xml", ''), ("book.opf", package), ("nav.xhtml", f'目次'), ("style.css", css), ("image.svg", svg)] + [(f"chapter-{i}.xhtml", html(i)) for i in range(500)] # Preserve the original short-tail corpus for end-clamping regressions. The # timed corpus makes all 13 heading destinations distinct at any viewport size, # without shortening the fixed traversal or counting unchanged positions. archive("standard-500-clamped.epub", epub_entries) archive("standard-500.epub", [(path, content + "\nsection:last-child{min-height:100vh}" if path == "style.css" else content) for path, content in epub_entries]) files = {str(path.relative_to(out)): {"bytes": path.stat().st_size, "sha256": hashlib.sha256(path.read_bytes()).hexdigest()} for path in sorted(out.rglob("*")) if path.is_file() and path.name not in ("manifest.json", "README.md")} files["standard-500.epub"]["purpose"] = "Timed navigation: 13 distinct heading destinations per chapter; final section minimum height is one viewport." files["standard-500-clamped.epub"]["purpose"] = "Original short-tail corpus, retained unchanged for end-clamping regressions and prior failed benchmark hashes." (out / "manifest.json").write_text(json.dumps({"origin": "DocView original generated test content; PDF embeds the OFL font described in ../pdf/extended/manifest.json", "limits": "500 PDF pages and 500 EPUB chapters, small reused images; not a 1 GiB scan corpus", "files": files}, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") print(json.dumps({name: info["bytes"] for name, info in files.items()}, ensure_ascii=False))