#!/usr/bin/env python3
"""Deterministic medium-size fixtures, assembled only from DocView test content.
These exercise page/chapter counts and local-resource navigation. They are not
1 GiB scan stress documents or a statistically representative book collection.
"""
from pathlib import Path
import hashlib
import json
import zipfile
from pypdf import PdfReader, PdfWriter
root = Path(__file__).resolve().parent
out = root / "fixtures/performance"
out.mkdir(parents=True, exist_ok=True)
pdf_root = root / "fixtures/pdf/extended"
sources = [PdfReader(pdf_root / name) for name in ("japanese-layout.pdf", "graphics-compositing.pdf")]
writer = PdfWriter()
for index in range(500):
source = sources[index % len(sources)]
writer.add_page(source.pages[index % len(source.pages)])
writer.add_metadata({"/Title": "DocView generated 500-page performance corpus", "/Author": "DocView tests"})
with (out / "standard-500.pdf").open("wb") as handle:
writer.write(handle)
# A separate navigation document keeps the established page/graphics corpus
# unchanged while providing 500 explicit, source-authored outline destinations.
heading_writer = PdfWriter(clone_from=out / "standard-500.pdf")
for index in range(500):
heading_writer.add_outline_item(f"Generated chapter {index + 1}", index)
with (out / "standard-headings-500.pdf").open("wb") as handle:
heading_writer.write(handle)
css = "body{font:18px sans-serif;line-height:1.6;margin:3em;max-width:55em}h1,h2{color:#165f69}img{width:240px}"
svg = ''
paragraph = "日本語の読み取りとローカル資源を確認します。This is original synthetic reading content."
def html(chapter, sections=12):
body = f'
第{chapter + 1}章
'
body += ''.join(f'
節 {chapter + 1}.{i + 1}
{paragraph} ({chapter}/{i})
' for i in range(sections))
return f'章{chapter + 1}{body}'
html_dir = out / "html"
html_dir.mkdir(exist_ok=True)
(html_dir / "index.html").write_text(html(0, 400), encoding="utf-8")
(html_dir / "style.css").write_text(css, encoding="utf-8")
(html_dir / "image.svg").write_text(svg, encoding="utf-8")
def archive(name, entries):
with zipfile.ZipFile(out / name, "w") as handle:
for path, content in entries:
info = zipfile.ZipInfo(path, (2026, 1, 1, 0, 0, 0))
info.create_system = 3
info.external_attr = 0o100600 << 16
info.compress_type = zipfile.ZIP_STORED if path == "mimetype" else zipfile.ZIP_DEFLATED
handle.writestr(info, content.encode("utf-8"))
archive("standard-html.zip", [("index.html", html(0, 400)), ("style.css", css), ("image.svg", svg)] +
[(f"chapter-{i}.html", html(i)) for i in range(1, 50)])
manifest = ''.join(f'' for i in range(500))
spine = ''.join(f'' for i in range(500))
nav = ''.join(f'
' for i in range(500))
package = f'urn:docview:generated:500生成した500章の資料ja{manifest}{spine}'
epub_entries = [("mimetype", "application/epub+zip"),
("META-INF/container.xml", ''),
("book.opf", package), ("nav.xhtml", f'目次'),
("style.css", css), ("image.svg", svg)] + [(f"chapter-{i}.xhtml", html(i)) for i in range(500)]
# Preserve the original short-tail corpus for end-clamping regressions. The
# timed corpus makes all 13 heading destinations distinct at any viewport size,
# without shortening the fixed traversal or counting unchanged positions.
archive("standard-500-clamped.epub", epub_entries)
archive("standard-500.epub", [(path, content + "\nsection:last-child{min-height:100vh}" if path == "style.css" else content)
for path, content in epub_entries])
files = {str(path.relative_to(out)): {"bytes": path.stat().st_size, "sha256": hashlib.sha256(path.read_bytes()).hexdigest()}
for path in sorted(out.rglob("*")) if path.is_file() and path.name not in ("manifest.json", "README.md")}
files["standard-500.epub"]["purpose"] = "Timed navigation: 13 distinct heading destinations per chapter; final section minimum height is one viewport."
files["standard-500-clamped.epub"]["purpose"] = "Original short-tail corpus, retained unchanged for end-clamping regressions and prior failed benchmark hashes."
(out / "manifest.json").write_text(json.dumps({"origin": "DocView original generated test content; PDF embeds the OFL font described in ../pdf/extended/manifest.json",
"limits": "500 PDF pages and 500 EPUB chapters, small reused images; not a 1 GiB scan corpus", "files": files}, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps({name: info["bytes"] for name, info in files.items()}, ensure_ascii=False))