initial commit

This commit is contained in:
2026-09-21 13:41:40 +09:00
commit 855c7328df
411 changed files with 85352 additions and 0 deletions
+178
View File
@@ -0,0 +1,178 @@
#!/usr/bin/env python3
"""Stream an original, deterministic scan-like PDF; no padding-only payload.
The default is 5,000 pages referencing all 1,280 unique 512x512 RGB images.
Each image is an uncompressed 768 KiB stream used by at least one page.
Only one image (plus its reusable template) is buffered at a time.
"""
import argparse
import hashlib
import json
import os
from pathlib import Path
import shutil
import time
ROOT = Path(__file__).resolve().parents[1]
WIDTH = HEIGHT = 512
IMAGE_BYTES = WIDTH * HEIGHT * 3
def raster_template():
raster = bytearray(IMAGE_BYTES)
for y in range(HEIGHT):
for x in range(WIDTH):
value = 244 + ((x * 13 + y * 7) % 12)
# Authored scan-like text blocks, row rules and page margins.
if 36 <= x < 476 and 80 <= y < 465:
row, within = divmod(y - 80, 22)
if within < 8 and x < 470 - (row % 4) * 53 and (x - 36) % 14 < 10:
value = 38 + ((x + y) % 20)
i = (y * WIDTH + x) * 3
raster[i:i + 3] = bytes((value, value, value))
return raster
def raster_for(template, index):
raster = template.copy()
# A 16-bit, visually legible image-ID barcode makes each image distinct.
for bit in range(16):
color = bytes((24, 77, 113)) if index & (1 << bit) else bytes((210, 222, 230))
stripe = color * 24
x = 48 + bit * 26
for y in range(28, 60):
i = (y * WIDTH + x) * 3
raster[i:i + len(stripe)] = stripe
return raster
def generate(path, pages=5000, images=1280):
if pages < 1 or images < 1 or images > pages or pages > 100000 or images > 65536:
raise ValueError("Require 1 <= images <= pages <= 100000 and images <= 65536")
path = Path(path).resolve()
path.parent.mkdir(parents=True, exist_ok=True)
estimated_bytes = images * IMAGE_BYTES + pages * 768 + 2 * 1024 * 1024
if estimated_bytes >= 10_000_000_000:
raise ValueError("The classic PDF cross-reference format is limited to ten-digit byte offsets")
free_before = shutil.disk_usage(path.parent).free
reserve = 512 * 1024 * 1024
if free_before < estimated_bytes + reserve:
raise RuntimeError("Insufficient free disk for the PDF plus a 512 MiB reserve")
if path.exists():
raise FileExistsError(f"Refusing to replace an existing file: {path}")
started = time.monotonic()
digest = hashlib.sha256()
image_first = 4
content_first = image_first + images
page_first = content_first + pages
object_count = page_first + pages - 1
offsets = [0] * (object_count + 1)
records = []
sample_pages = sorted(set([0, min(97, pages - 1), min(images - 1, pages - 1), pages // 2, pages - 1]))
page_records = []
template = raster_template()
try:
with path.open("xb", buffering=1024 * 1024) as out:
def write(data):
out.write(data)
digest.update(data)
def begin(number):
offsets[number] = out.tell()
write(f"{number} 0 obj\n".encode())
def object_(number, body):
begin(number)
write(body + b"\nendobj\n")
write(b"%PDF-1.7\n%\xe2\xe3\xcf\xd3\n")
object_(1, b"<< /Type /Catalog /Pages 2 0 R >>")
kids = " ".join(f"{page_first + i} 0 R" for i in range(pages))
object_(2, f"<< /Type /Pages /Count {pages} /Kids [{kids}] >>".encode())
object_(3, b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>")
for index in range(images):
raster = raster_for(template, index)
begin(image_first + index)
write((f"<< /Type /XObject /Subtype /Image /Width {WIDTH} /Height {HEIGHT} "
f"/ColorSpace /DeviceRGB /BitsPerComponent 8 /Length {IMAGE_BYTES} >>\nstream\n").encode())
stream_offset = out.tell()
write(raster)
write(b"\nendstream\nendobj\n")
records.append({"index": index, "object": image_first + index,
"streamOffset": stream_offset, "bytes": len(raster),
"sha256": hashlib.sha256(raster).hexdigest()})
del raster
for page in range(pages):
image = page % images
marker = f"DOCVIEW STRESS PAGE {page + 1:05d} OF {pages:05d} IMAGE {image:04d}"
stream = (f"q 512 0 0 512 44 180 cm /Scan Do Q\n"
f"0.1 0.3 0.6 rg 24 748 12 12 re f\n"
f"0.8 0.2 0.3 rg 564 24 12 12 re f\n"
f"0 g BT /F1 14 Tf 44 752 Td ({marker}) Tj ET\n"
f"BT /F1 11 Tf 44 142 Td (Original synthetic raster; no hidden padding.) Tj ET\n"
f"BT /F1 11 Tf 44 120 Td (Raster image {image:04d}; 512 x 512 RGB pixels.) Tj ET\n").encode()
object_(content_first + page, f"<< /Length {len(stream)} >>\nstream\n".encode() + stream + b"endstream")
if page in sample_pages:
page_records.append({"pageIndex": page, "physicalPage": page + 1, "imageIndex": image,
"pageObject": page_first + page, "contentObject": content_first + page,
"marker": marker, "textBaselinePt": [44, 752]})
for page in range(pages):
object_(page_first + page,
(f"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 600 800] "
f"/Resources << /Font << /F1 3 0 R >> /XObject << /Scan {image_first + page % images} 0 R >> >> "
f"/Contents {content_first + page} 0 R >>").encode())
xref = out.tell()
write(f"xref\n0 {object_count + 1}\n0000000000 65535 f \n".encode())
for offset in offsets[1:]:
write(f"{offset:010d} 00000 n \n".encode())
write((f"trailer\n<< /Size {object_count + 1} /Root 1 0 R >>\n"
f"startxref\n{xref}\n%%EOF\n").encode())
out.flush()
os.fsync(out.fileno())
except BaseException:
path.unlink(missing_ok=True)
raise
return {"schemaVersion": 1, "generator": "generate_stress_pdf.py", "source": str(path),
"sha256": digest.hexdigest(), "sizeBytes": path.stat().st_size,
"pageCount": pages, "uniqueImageCount": images, "rasterPayloadBytes": images * IMAGE_BYTES,
"generationSeconds": time.monotonic() - started, "freeDiskBeforeBytes": free_before,
"maxRasterBufferBytes": IMAGE_BYTES * 2, "pageSizePt": [600, 800],
"imageRectanglePt": [44, 180, 556, 692],
"markerRectanglesPt": [[24, 748, 36, 760], [564, 24, 576, 36]],
"imageUse": "pageIndex % uniqueImageCount; every image is used, streams have no filter",
"images": records, "samplePages": page_records, "xrefOffset": xref,
"origin": "Self-authored raster texture, block glyphs, barcode and page markers; no external content"}
def validate_source(path, manifest):
path = Path(path)
if path.stat().st_size != manifest["sizeBytes"]:
raise AssertionError("File size mismatch")
images = manifest["images"]
if len({image["sha256"] for image in images}) != len(images):
raise AssertionError("Raster images are not unique")
checked = sorted(set([0, len(images) // 2, len(images) - 1]))
with path.open("rb") as source:
source.seek(manifest["xrefOffset"])
if source.read(5) != b"xref\n":
raise AssertionError("Cross-reference offset mismatch")
for index in checked:
image = images[index]
source.seek(image["streamOffset"])
if hashlib.sha256(source.read(image["bytes"])).hexdigest() != image["sha256"]:
raise AssertionError("Raster stream mismatch")
return {"sampledImageIndices": checked, "uniqueRasterDigests": len(images),
"sizeAndXrefValid": True}
if __name__ == "__main__":
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--output", type=Path, default=ROOT / "build/stress/scan-5000.pdf")
parser.add_argument("--pages", type=int, default=5000)
parser.add_argument("--images", type=int, default=1280)
args = parser.parse_args()
manifest = generate(args.output, args.pages, args.images)
manifest["validation"] = validate_source(args.output, manifest)
manifest_path = args.output.with_suffix(".json")
manifest_path.write_text(json.dumps(manifest, indent=2) + "\n", encoding="utf-8")
print(json.dumps({key: value for key, value in manifest.items() if key != "images"}, indent=2))