81 lines
5.9 KiB
Python
81 lines
5.9 KiB
Python
#!/usr/bin/env python3
|
|
"""Deterministic medium-size fixtures, assembled only from DocView test content.
|
|
|
|
These exercise page/chapter counts and local-resource navigation. They are not
|
|
1 GiB scan stress documents or a statistically representative book collection.
|
|
"""
|
|
from pathlib import Path
|
|
import hashlib
|
|
import json
|
|
import zipfile
|
|
from pypdf import PdfReader, PdfWriter
|
|
|
|
root = Path(__file__).resolve().parent
|
|
out = root / "fixtures/performance"
|
|
out.mkdir(parents=True, exist_ok=True)
|
|
pdf_root = root / "fixtures/pdf/extended"
|
|
sources = [PdfReader(pdf_root / name) for name in ("japanese-layout.pdf", "graphics-compositing.pdf")]
|
|
writer = PdfWriter()
|
|
for index in range(500):
|
|
source = sources[index % len(sources)]
|
|
writer.add_page(source.pages[index % len(source.pages)])
|
|
writer.add_metadata({"/Title": "DocView generated 500-page performance corpus", "/Author": "DocView tests"})
|
|
with (out / "standard-500.pdf").open("wb") as handle:
|
|
writer.write(handle)
|
|
|
|
# A separate navigation document keeps the established page/graphics corpus
|
|
# unchanged while providing 500 explicit, source-authored outline destinations.
|
|
heading_writer = PdfWriter(clone_from=out / "standard-500.pdf")
|
|
for index in range(500):
|
|
heading_writer.add_outline_item(f"Generated chapter {index + 1}", index)
|
|
with (out / "standard-headings-500.pdf").open("wb") as handle:
|
|
heading_writer.write(handle)
|
|
|
|
css = "body{font:18px sans-serif;line-height:1.6;margin:3em;max-width:55em}h1,h2{color:#165f69}img{width:240px}"
|
|
svg = '<svg xmlns="http://www.w3.org/2000/svg" width="480" height="240"><rect width="480" height="240" fill="#dae8ed"/><circle cx="240" cy="120" r="85" fill="#247b84"/></svg>'
|
|
paragraph = "日本語の読み取りとローカル資源を確認します。This is original synthetic reading content."
|
|
|
|
def html(chapter, sections=12):
|
|
body = f'<h1 id="chapter">第{chapter + 1}章</h1>'
|
|
body += ''.join(f'<section><h2 id="section-{i}">節 {chapter + 1}.{i + 1}</h2><p>{paragraph} ({chapter}/{i})</p><img src="image.svg" alt="検証用の円"/></section>' for i in range(sections))
|
|
return f'<!DOCTYPE html><html xmlns="http://www.w3.org/1999/xhtml" lang="ja"><head><meta charset="utf-8"/><title>章{chapter + 1}</title><link rel="stylesheet" href="style.css"/></head><body>{body}</body></html>'
|
|
|
|
html_dir = out / "html"
|
|
html_dir.mkdir(exist_ok=True)
|
|
(html_dir / "index.html").write_text(html(0, 400), encoding="utf-8")
|
|
(html_dir / "style.css").write_text(css, encoding="utf-8")
|
|
(html_dir / "image.svg").write_text(svg, encoding="utf-8")
|
|
|
|
def archive(name, entries):
|
|
with zipfile.ZipFile(out / name, "w") as handle:
|
|
for path, content in entries:
|
|
info = zipfile.ZipInfo(path, (2026, 1, 1, 0, 0, 0))
|
|
info.create_system = 3
|
|
info.external_attr = 0o100600 << 16
|
|
info.compress_type = zipfile.ZIP_STORED if path == "mimetype" else zipfile.ZIP_DEFLATED
|
|
handle.writestr(info, content.encode("utf-8"))
|
|
|
|
archive("standard-html.zip", [("index.html", html(0, 400)), ("style.css", css), ("image.svg", svg)] +
|
|
[(f"chapter-{i}.html", html(i)) for i in range(1, 50)])
|
|
manifest = ''.join(f'<item id="c{i}" href="chapter-{i}.xhtml" media-type="application/xhtml+xml"/>' for i in range(500))
|
|
spine = ''.join(f'<itemref idref="c{i}"/>' for i in range(500))
|
|
nav = ''.join(f'<li><a href="chapter-{i}.xhtml#chapter">第{i + 1}章</a></li>' for i in range(500))
|
|
package = f'<package xmlns="http://www.idpf.org/2007/opf" version="3.0" unique-identifier="id"><metadata xmlns:dc="http://purl.org/dc/elements/1.1/"><dc:identifier id="id">urn:docview:generated:500</dc:identifier><dc:title>生成した500章の資料</dc:title><dc:language>ja</dc:language></metadata><manifest>{manifest}<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml" properties="nav"/><item id="css" href="style.css" media-type="text/css"/><item id="img" href="image.svg" media-type="image/svg+xml"/></manifest><spine>{spine}</spine></package>'
|
|
epub_entries = [("mimetype", "application/epub+zip"),
|
|
("META-INF/container.xml", '<container xmlns="urn:oasis:names:tc:opendocument:xmlns:container" version="1.0"><rootfiles><rootfile full-path="book.opf" media-type="application/oebps-package+xml"/></rootfiles></container>'),
|
|
("book.opf", package), ("nav.xhtml", f'<html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops"><head><title>目次</title></head><body><nav epub:type="toc"><ol>{nav}</ol></nav></body></html>'),
|
|
("style.css", css), ("image.svg", svg)] + [(f"chapter-{i}.xhtml", html(i)) for i in range(500)]
|
|
# Preserve the original short-tail corpus for end-clamping regressions. The
|
|
# timed corpus makes all 13 heading destinations distinct at any viewport size,
|
|
# without shortening the fixed traversal or counting unchanged positions.
|
|
archive("standard-500-clamped.epub", epub_entries)
|
|
archive("standard-500.epub", [(path, content + "\nsection:last-child{min-height:100vh}" if path == "style.css" else content)
|
|
for path, content in epub_entries])
|
|
files = {str(path.relative_to(out)): {"bytes": path.stat().st_size, "sha256": hashlib.sha256(path.read_bytes()).hexdigest()}
|
|
for path in sorted(out.rglob("*")) if path.is_file() and path.name not in ("manifest.json", "README.md")}
|
|
files["standard-500.epub"]["purpose"] = "Timed navigation: 13 distinct heading destinations per chapter; final section minimum height is one viewport."
|
|
files["standard-500-clamped.epub"]["purpose"] = "Original short-tail corpus, retained unchanged for end-clamping regressions and prior failed benchmark hashes."
|
|
(out / "manifest.json").write_text(json.dumps({"origin": "DocView original generated test content; PDF embeds the OFL font described in ../pdf/extended/manifest.json",
|
|
"limits": "500 PDF pages and 500 EPUB chapters, small reused images; not a 1 GiB scan corpus", "files": files}, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
print(json.dumps({name: info["bytes"] for name, info in files.items()}, ensure_ascii=False))
|