initial commit

This commit is contained in:
2026-09-21 13:41:40 +09:00
commit 855c7328df
411 changed files with 85352 additions and 0 deletions
+80
View File
@@ -0,0 +1,80 @@
#!/usr/bin/env python3
"""Deterministic medium-size fixtures, assembled only from DocView test content.
These exercise page/chapter counts and local-resource navigation. They are not
1 GiB scan stress documents or a statistically representative book collection.
"""
from pathlib import Path
import hashlib
import json
import zipfile
from pypdf import PdfReader, PdfWriter
root = Path(__file__).resolve().parent
out = root / "fixtures/performance"
out.mkdir(parents=True, exist_ok=True)
pdf_root = root / "fixtures/pdf/extended"
sources = [PdfReader(pdf_root / name) for name in ("japanese-layout.pdf", "graphics-compositing.pdf")]
writer = PdfWriter()
for index in range(500):
source = sources[index % len(sources)]
writer.add_page(source.pages[index % len(source.pages)])
writer.add_metadata({"/Title": "DocView generated 500-page performance corpus", "/Author": "DocView tests"})
with (out / "standard-500.pdf").open("wb") as handle:
writer.write(handle)
# A separate navigation document keeps the established page/graphics corpus
# unchanged while providing 500 explicit, source-authored outline destinations.
heading_writer = PdfWriter(clone_from=out / "standard-500.pdf")
for index in range(500):
heading_writer.add_outline_item(f"Generated chapter {index + 1}", index)
with (out / "standard-headings-500.pdf").open("wb") as handle:
heading_writer.write(handle)
css = "body{font:18px sans-serif;line-height:1.6;margin:3em;max-width:55em}h1,h2{color:#165f69}img{width:240px}"
svg = '<svg xmlns="http://www.w3.org/2000/svg" width="480" height="240"><rect width="480" height="240" fill="#dae8ed"/><circle cx="240" cy="120" r="85" fill="#247b84"/></svg>'
paragraph = "日本語の読み取りとローカル資源を確認します。This is original synthetic reading content."
def html(chapter, sections=12):
body = f'<h1 id="chapter">第{chapter + 1}章</h1>'
body += ''.join(f'<section><h2 id="section-{i}">節 {chapter + 1}.{i + 1}</h2><p>{paragraph} ({chapter}/{i})</p><img src="image.svg" alt="検証用の円"/></section>' for i in range(sections))
return f'<!DOCTYPE html><html xmlns="http://www.w3.org/1999/xhtml" lang="ja"><head><meta charset="utf-8"/><title>章{chapter + 1}</title><link rel="stylesheet" href="style.css"/></head><body>{body}</body></html>'
html_dir = out / "html"
html_dir.mkdir(exist_ok=True)
(html_dir / "index.html").write_text(html(0, 400), encoding="utf-8")
(html_dir / "style.css").write_text(css, encoding="utf-8")
(html_dir / "image.svg").write_text(svg, encoding="utf-8")
def archive(name, entries):
with zipfile.ZipFile(out / name, "w") as handle:
for path, content in entries:
info = zipfile.ZipInfo(path, (2026, 1, 1, 0, 0, 0))
info.create_system = 3
info.external_attr = 0o100600 << 16
info.compress_type = zipfile.ZIP_STORED if path == "mimetype" else zipfile.ZIP_DEFLATED
handle.writestr(info, content.encode("utf-8"))
archive("standard-html.zip", [("index.html", html(0, 400)), ("style.css", css), ("image.svg", svg)] +
[(f"chapter-{i}.html", html(i)) for i in range(1, 50)])
manifest = ''.join(f'<item id="c{i}" href="chapter-{i}.xhtml" media-type="application/xhtml+xml"/>' for i in range(500))
spine = ''.join(f'<itemref idref="c{i}"/>' for i in range(500))
nav = ''.join(f'<li><a href="chapter-{i}.xhtml#chapter">第{i + 1}章</a></li>' for i in range(500))
package = f'<package xmlns="http://www.idpf.org/2007/opf" version="3.0" unique-identifier="id"><metadata xmlns:dc="http://purl.org/dc/elements/1.1/"><dc:identifier id="id">urn:docview:generated:500</dc:identifier><dc:title>生成した500章の資料</dc:title><dc:language>ja</dc:language></metadata><manifest>{manifest}<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml" properties="nav"/><item id="css" href="style.css" media-type="text/css"/><item id="img" href="image.svg" media-type="image/svg+xml"/></manifest><spine>{spine}</spine></package>'
epub_entries = [("mimetype", "application/epub+zip"),
("META-INF/container.xml", '<container xmlns="urn:oasis:names:tc:opendocument:xmlns:container" version="1.0"><rootfiles><rootfile full-path="book.opf" media-type="application/oebps-package+xml"/></rootfiles></container>'),
("book.opf", package), ("nav.xhtml", f'<html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops"><head><title>目次</title></head><body><nav epub:type="toc"><ol>{nav}</ol></nav></body></html>'),
("style.css", css), ("image.svg", svg)] + [(f"chapter-{i}.xhtml", html(i)) for i in range(500)]
# Preserve the original short-tail corpus for end-clamping regressions. The
# timed corpus makes all 13 heading destinations distinct at any viewport size,
# without shortening the fixed traversal or counting unchanged positions.
archive("standard-500-clamped.epub", epub_entries)
archive("standard-500.epub", [(path, content + "\nsection:last-child{min-height:100vh}" if path == "style.css" else content)
for path, content in epub_entries])
files = {str(path.relative_to(out)): {"bytes": path.stat().st_size, "sha256": hashlib.sha256(path.read_bytes()).hexdigest()}
for path in sorted(out.rglob("*")) if path.is_file() and path.name not in ("manifest.json", "README.md")}
files["standard-500.epub"]["purpose"] = "Timed navigation: 13 distinct heading destinations per chapter; final section minimum height is one viewport."
files["standard-500-clamped.epub"]["purpose"] = "Original short-tail corpus, retained unchanged for end-clamping regressions and prior failed benchmark hashes."
(out / "manifest.json").write_text(json.dumps({"origin": "DocView original generated test content; PDF embeds the OFL font described in ../pdf/extended/manifest.json",
"limits": "500 PDF pages and 500 EPUB chapters, small reused images; not a 1 GiB scan corpus", "files": files}, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps({name: info["bytes"] for name, info in files.items()}, ensure_ascii=False))