63 lines
3.7 KiB
Python
63 lines
3.7 KiB
Python
#!/usr/bin/env python3
|
|
"""Production worker vs independent Poppler for unembedded Japanese CID H/V."""
|
|
import argparse
|
|
import hashlib
|
|
import json
|
|
import os
|
|
from pathlib import Path
|
|
import subprocess
|
|
from PIL import Image, ImageChops, ImageDraw, ImageStat
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("--build-dir", type=Path, default=ROOT / os.environ.get("DOCVIEW_BUILD_DIR", "build"))
|
|
parser.add_argument("--output", type=Path, default=ROOT / "tests/results/pdf-fonts")
|
|
args = parser.parse_args()
|
|
BUILD = args.build_dir.resolve()
|
|
CORPUS = ROOT / "tests/fixtures/pdf/fonts"
|
|
OUT = args.output.resolve()
|
|
OUT.mkdir(parents=True, exist_ok=True)
|
|
specification = json.loads((CORPUS / "manifest.json").read_text())["documents"][0]
|
|
source = CORPUS / specification["file"]
|
|
assert hashlib.sha256(source.read_bytes()).hexdigest() == specification["sha256"]
|
|
subprocess.run([str(BUILD / "pdf-worker-evidence"), str(source), str(OUT / "worker")], check=True)
|
|
poppler = os.environ.get("DOCVIEW_PDFTOPPM", "/usr/bin/pdftoppm")
|
|
subprocess.run([poppler, "-r", "144", "-png", str(source), str(OUT / "poppler")], check=True)
|
|
subprocess.run(["/usr/bin/pdftotext", "-layout", str(source), str(OUT / "poppler-text.txt")], check=True)
|
|
fonts = subprocess.check_output(["/usr/bin/pdffonts", str(source)], text=True)
|
|
(OUT / "font-list.txt").write_text(fonts)
|
|
metadata = json.loads((OUT / "worker/metadata.json").read_text())
|
|
report = {
|
|
"reference_status": "Independent renderer comparison; no human-approved golden image or comprehensive CID/CMap acceptance",
|
|
"fixture_sha256": specification["sha256"],
|
|
"worker_sha256": hashlib.sha256((BUILD / "docview-pdf-worker").read_bytes()).hexdigest(),
|
|
"broker_sha256": hashlib.sha256((BUILD / "libdocview_font_broker.a").read_bytes()).hexdigest(),
|
|
"transport": metadata["transport"],
|
|
"poppler_version": subprocess.run([poppler, "-v"], text=True, capture_output=True).stderr.splitlines()[0],
|
|
"font_selections": metadata["fontSelections"],
|
|
"font_bytes_transferred": metadata["fontBytesTransferred"],
|
|
"japanese_search_matches_first_page": len(metadata.get("japaneseSearch", {}).get("matches", [])),
|
|
"pages": [],
|
|
}
|
|
for page in (1, 2):
|
|
a = Image.open(OUT / "worker" / f"page-{page}.png").convert("RGB")
|
|
b = Image.open(OUT / f"poppler-{page}.png").convert("RGB")
|
|
assert a.size == b.size == (1200, 1520)
|
|
difference = ImageChops.difference(a, b)
|
|
difference.save(OUT / f"difference-{page}.png")
|
|
histogram = ImageChops.lighter(ImageChops.lighter(difference.getchannel("R"), difference.getchannel("G")), difference.getchannel("B")).histogram()
|
|
report["pages"].append({"page": page, "size": list(a.size), "dimensions_match_source": True,
|
|
"equal_pixels": histogram[0], "pixels_max_delta_gt_16": sum(histogram[17:]),
|
|
"mean_absolute_channel_error": sum(ImageStat.Stat(difference).mean) / 3})
|
|
comparison = Image.new("RGB", (1200, 550), "#eeeeee")
|
|
draw = ImageDraw.Draw(comparison)
|
|
for col, (title, picture) in enumerate((("Sandboxed DocView / PDFium", a), ("Independent Poppler", b), ("Absolute RGB difference", difference))):
|
|
draw.text((col * 400 + 8, 10), title, fill="black")
|
|
picture.thumbnail((390, 500)); comparison.paste(picture, (col * 400 + 5, 35))
|
|
comparison.save(OUT / f"comparison-{page}.png")
|
|
assert report["font_bytes_transferred"] > 0
|
|
assert any(item.get("charset") == 128 and item.get("found") for item in report["font_selections"])
|
|
assert report["japanese_search_matches_first_page"] == 2
|
|
(OUT / "comparison.json").write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n")
|
|
print(json.dumps(report, ensure_ascii=False, indent=2))
|