179 lines
8.6 KiB
Python
179 lines
8.6 KiB
Python
#!/usr/bin/env python3
|
|
"""Stream an original, deterministic scan-like PDF; no padding-only payload.
|
|
|
|
The default is 5,000 pages referencing all 1,280 unique 512x512 RGB images.
|
|
Each image is an uncompressed 768 KiB stream used by at least one page.
|
|
Only one image (plus its reusable template) is buffered at a time.
|
|
"""
|
|
import argparse
|
|
import hashlib
|
|
import json
|
|
import os
|
|
from pathlib import Path
|
|
import shutil
|
|
import time
|
|
|
|
ROOT = Path(__file__).resolve().parents[1]
|
|
WIDTH = HEIGHT = 512
|
|
IMAGE_BYTES = WIDTH * HEIGHT * 3
|
|
|
|
|
|
def raster_template():
|
|
raster = bytearray(IMAGE_BYTES)
|
|
for y in range(HEIGHT):
|
|
for x in range(WIDTH):
|
|
value = 244 + ((x * 13 + y * 7) % 12)
|
|
# Authored scan-like text blocks, row rules and page margins.
|
|
if 36 <= x < 476 and 80 <= y < 465:
|
|
row, within = divmod(y - 80, 22)
|
|
if within < 8 and x < 470 - (row % 4) * 53 and (x - 36) % 14 < 10:
|
|
value = 38 + ((x + y) % 20)
|
|
i = (y * WIDTH + x) * 3
|
|
raster[i:i + 3] = bytes((value, value, value))
|
|
return raster
|
|
|
|
|
|
def raster_for(template, index):
|
|
raster = template.copy()
|
|
# A 16-bit, visually legible image-ID barcode makes each image distinct.
|
|
for bit in range(16):
|
|
color = bytes((24, 77, 113)) if index & (1 << bit) else bytes((210, 222, 230))
|
|
stripe = color * 24
|
|
x = 48 + bit * 26
|
|
for y in range(28, 60):
|
|
i = (y * WIDTH + x) * 3
|
|
raster[i:i + len(stripe)] = stripe
|
|
return raster
|
|
|
|
|
|
def generate(path, pages=5000, images=1280):
|
|
if pages < 1 or images < 1 or images > pages or pages > 100000 or images > 65536:
|
|
raise ValueError("Require 1 <= images <= pages <= 100000 and images <= 65536")
|
|
path = Path(path).resolve()
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
estimated_bytes = images * IMAGE_BYTES + pages * 768 + 2 * 1024 * 1024
|
|
if estimated_bytes >= 10_000_000_000:
|
|
raise ValueError("The classic PDF cross-reference format is limited to ten-digit byte offsets")
|
|
free_before = shutil.disk_usage(path.parent).free
|
|
reserve = 512 * 1024 * 1024
|
|
if free_before < estimated_bytes + reserve:
|
|
raise RuntimeError("Insufficient free disk for the PDF plus a 512 MiB reserve")
|
|
if path.exists():
|
|
raise FileExistsError(f"Refusing to replace an existing file: {path}")
|
|
started = time.monotonic()
|
|
digest = hashlib.sha256()
|
|
image_first = 4
|
|
content_first = image_first + images
|
|
page_first = content_first + pages
|
|
object_count = page_first + pages - 1
|
|
offsets = [0] * (object_count + 1)
|
|
records = []
|
|
sample_pages = sorted(set([0, min(97, pages - 1), min(images - 1, pages - 1), pages // 2, pages - 1]))
|
|
page_records = []
|
|
template = raster_template()
|
|
try:
|
|
with path.open("xb", buffering=1024 * 1024) as out:
|
|
def write(data):
|
|
out.write(data)
|
|
digest.update(data)
|
|
|
|
def begin(number):
|
|
offsets[number] = out.tell()
|
|
write(f"{number} 0 obj\n".encode())
|
|
|
|
def object_(number, body):
|
|
begin(number)
|
|
write(body + b"\nendobj\n")
|
|
|
|
write(b"%PDF-1.7\n%\xe2\xe3\xcf\xd3\n")
|
|
object_(1, b"<< /Type /Catalog /Pages 2 0 R >>")
|
|
kids = " ".join(f"{page_first + i} 0 R" for i in range(pages))
|
|
object_(2, f"<< /Type /Pages /Count {pages} /Kids [{kids}] >>".encode())
|
|
object_(3, b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>")
|
|
for index in range(images):
|
|
raster = raster_for(template, index)
|
|
begin(image_first + index)
|
|
write((f"<< /Type /XObject /Subtype /Image /Width {WIDTH} /Height {HEIGHT} "
|
|
f"/ColorSpace /DeviceRGB /BitsPerComponent 8 /Length {IMAGE_BYTES} >>\nstream\n").encode())
|
|
stream_offset = out.tell()
|
|
write(raster)
|
|
write(b"\nendstream\nendobj\n")
|
|
records.append({"index": index, "object": image_first + index,
|
|
"streamOffset": stream_offset, "bytes": len(raster),
|
|
"sha256": hashlib.sha256(raster).hexdigest()})
|
|
del raster
|
|
for page in range(pages):
|
|
image = page % images
|
|
marker = f"DOCVIEW STRESS PAGE {page + 1:05d} OF {pages:05d} IMAGE {image:04d}"
|
|
stream = (f"q 512 0 0 512 44 180 cm /Scan Do Q\n"
|
|
f"0.1 0.3 0.6 rg 24 748 12 12 re f\n"
|
|
f"0.8 0.2 0.3 rg 564 24 12 12 re f\n"
|
|
f"0 g BT /F1 14 Tf 44 752 Td ({marker}) Tj ET\n"
|
|
f"BT /F1 11 Tf 44 142 Td (Original synthetic raster; no hidden padding.) Tj ET\n"
|
|
f"BT /F1 11 Tf 44 120 Td (Raster image {image:04d}; 512 x 512 RGB pixels.) Tj ET\n").encode()
|
|
object_(content_first + page, f"<< /Length {len(stream)} >>\nstream\n".encode() + stream + b"endstream")
|
|
if page in sample_pages:
|
|
page_records.append({"pageIndex": page, "physicalPage": page + 1, "imageIndex": image,
|
|
"pageObject": page_first + page, "contentObject": content_first + page,
|
|
"marker": marker, "textBaselinePt": [44, 752]})
|
|
for page in range(pages):
|
|
object_(page_first + page,
|
|
(f"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 600 800] "
|
|
f"/Resources << /Font << /F1 3 0 R >> /XObject << /Scan {image_first + page % images} 0 R >> >> "
|
|
f"/Contents {content_first + page} 0 R >>").encode())
|
|
xref = out.tell()
|
|
write(f"xref\n0 {object_count + 1}\n0000000000 65535 f \n".encode())
|
|
for offset in offsets[1:]:
|
|
write(f"{offset:010d} 00000 n \n".encode())
|
|
write((f"trailer\n<< /Size {object_count + 1} /Root 1 0 R >>\n"
|
|
f"startxref\n{xref}\n%%EOF\n").encode())
|
|
out.flush()
|
|
os.fsync(out.fileno())
|
|
except BaseException:
|
|
path.unlink(missing_ok=True)
|
|
raise
|
|
return {"schemaVersion": 1, "generator": "generate_stress_pdf.py", "source": str(path),
|
|
"sha256": digest.hexdigest(), "sizeBytes": path.stat().st_size,
|
|
"pageCount": pages, "uniqueImageCount": images, "rasterPayloadBytes": images * IMAGE_BYTES,
|
|
"generationSeconds": time.monotonic() - started, "freeDiskBeforeBytes": free_before,
|
|
"maxRasterBufferBytes": IMAGE_BYTES * 2, "pageSizePt": [600, 800],
|
|
"imageRectanglePt": [44, 180, 556, 692],
|
|
"markerRectanglesPt": [[24, 748, 36, 760], [564, 24, 576, 36]],
|
|
"imageUse": "pageIndex % uniqueImageCount; every image is used, streams have no filter",
|
|
"images": records, "samplePages": page_records, "xrefOffset": xref,
|
|
"origin": "Self-authored raster texture, block glyphs, barcode and page markers; no external content"}
|
|
|
|
|
|
def validate_source(path, manifest):
|
|
path = Path(path)
|
|
if path.stat().st_size != manifest["sizeBytes"]:
|
|
raise AssertionError("File size mismatch")
|
|
images = manifest["images"]
|
|
if len({image["sha256"] for image in images}) != len(images):
|
|
raise AssertionError("Raster images are not unique")
|
|
checked = sorted(set([0, len(images) // 2, len(images) - 1]))
|
|
with path.open("rb") as source:
|
|
source.seek(manifest["xrefOffset"])
|
|
if source.read(5) != b"xref\n":
|
|
raise AssertionError("Cross-reference offset mismatch")
|
|
for index in checked:
|
|
image = images[index]
|
|
source.seek(image["streamOffset"])
|
|
if hashlib.sha256(source.read(image["bytes"])).hexdigest() != image["sha256"]:
|
|
raise AssertionError("Raster stream mismatch")
|
|
return {"sampledImageIndices": checked, "uniqueRasterDigests": len(images),
|
|
"sizeAndXrefValid": True}
|
|
|
|
|
|
if __name__ == "__main__":
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("--output", type=Path, default=ROOT / "build/stress/scan-5000.pdf")
|
|
parser.add_argument("--pages", type=int, default=5000)
|
|
parser.add_argument("--images", type=int, default=1280)
|
|
args = parser.parse_args()
|
|
manifest = generate(args.output, args.pages, args.images)
|
|
manifest["validation"] = validate_source(args.output, manifest)
|
|
manifest_path = args.output.with_suffix(".json")
|
|
manifest_path.write_text(json.dumps(manifest, indent=2) + "\n", encoding="utf-8")
|
|
print(json.dumps({key: value for key, value in manifest.items() if key != "images"}, indent=2))
|