Files
docview/tests/generate_render_review.py
2026-09-21 13:41:40 +09:00

241 lines
15 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Assemble existing rendering evidence for human review; never grant approval."""
import argparse
import hashlib
import json
import os
from pathlib import Path
import struct
from urllib.parse import quote
ROOT = Path(__file__).resolve().parents[1]
TESTS = ROOT / 'tests'
RESULTS = TESTS / 'results'
OUTPUT = RESULTS / 'render-review'
def sha(path):
with path.open('rb') as handle:
return hashlib.file_digest(handle, 'sha256').hexdigest()
def asset(path, expected=None, label=None):
path = path.resolve(strict=True)
if not path.is_relative_to(TESTS):
raise ValueError(f'Evidence outside tests directory: {path}')
digest = sha(path)
if expected is not None and expected != digest:
raise ValueError(f'Stale evidence: {path}')
value = {'url': quote(os.path.relpath(path, OUTPUT)), 'sha256': digest}
if label:
value['label'] = label
if path.suffix == '.png':
with path.open('rb') as handle:
header = handle.read(24)
if header[:8] != b'\x89PNG\r\n\x1a\n' or header[12:16] != b'IHDR':
raise ValueError(f'Invalid PNG header: {path}')
value['size'] = list(struct.unpack('>II', header[16:24]))
return value
def main():
global RESULTS, OUTPUT
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--results-root', type=Path, default=RESULTS,
help='Parent of the six comparison directories; gallery is written to render-review below it.')
parser.add_argument('--build-dir', type=Path, default=ROOT / 'build',
help='Build whose PDF worker must match all six comparison reports.')
parser.add_argument('--output', type=Path,
help='Gallery directory; defaults to RESULTS_ROOT/render-review.')
args = parser.parse_args()
RESULTS = args.results_root.resolve()
OUTPUT = args.output.resolve() if args.output else RESULTS / 'render-review'
worker = sha(args.build_dir.resolve() / 'docview-pdf-worker')
comparisons = {}
inputs = []
for name in ['pdf', 'pdf-extended', 'pdf-fonts', 'link-borders', 'pdf-icc', 'font-collection']:
path = RESULTS / name / 'comparison.json'
values = json.loads(path.read_text())
if values['worker_sha256'] != worker:
raise ValueError(f'Comparison does not match current worker: {name}')
comparisons[name] = values
inputs.append(asset(path))
cases = []
def add(id_, group, title, source, source_hash, left, right, diff, size, checks,
notes, measurement, independent=True):
entry = {
'id': id_, 'group': group, 'title': title,
'source': asset(source, source_hash),
'left': asset(left, label='DocView / PDFium' if independent else 'PDFium:個別TTF'),
'right': asset(right, label='Poppler(独立レンダラー)' if independent else 'PDFium:TTC'),
'difference': asset(diff) if diff else None,
'independentReference': independent,
'checks': checks, 'notes': notes,
'metrics': f'{size[0]} × {size[1]} px / RGB平均絶対差 {measurement:.4f} / 255(合否閾値ではありません)',
}
for key in ['left', 'right', 'difference']:
if entry[key] and entry[key]['size'] != size:
raise ValueError(f'Dimension mismatch: {id_}/{key}')
cases.append(entry)
basic = comparisons['pdf']
add('basic-page-1', '基本描画', '本文・リンク・保存済みフォーム外観',
TESTS / 'fixtures/pdf/navigation.pdf', basic['fixture_sha256'],
RESULTS / 'pdf/pdfium.png', RESULTS / 'pdf/poppler.png', RESULTS / 'pdf/difference.png',
basic['size'], ['Heading One、Needle、Second MCIDと黒い矩形が見える。',
'保存されたフォーム外観は赤い矩形。下部のリンク枠が見える。'],
'文字のアンチエイリアスと、保存外観のないText注釈のアイコンは描画器により異なります。'
'フィールド値 Saved の文字表示はこの資料の期待値ではありません。第2ページはこの比較に含みません。',
basic['mean_absolute_channel_error'])
extended_checks = {
'japanese-layout': [],
'graphics-compositing': [
'RGB/RGBA画像、円形クリップ、グラデーション、透明な重なり、CMYK色見本が表示される。',
'50%透明合成の3点は生成時の期待色に対して両描画器とも各チャンネル差2以内。'],
'page-geometry': [
'赤・緑・青・橙の4隅のマーカーがページの回転に応じた位置にある。',
'CropBoxの外にあるピンク色の余白が表示されない。'],
'unembedded-font': ['Standard-14のテキストが欠落せず、行と字形を確認できる。'],
'icc-linear-rgb': [
'上段ICCベクターと下段ICC画像は、中段DeviceRGBより明るい。',
'線形0.5の灰色は約188、中段DeviceRGBは128。'],
}
extended_notes = {
'japanese-layout': '埋込BIZ UDPGothicと明示的な文字座標の検証です。この縦列はnative縦書きシェーピングの試験ではありません。',
'graphics-compositing': '文字や画像の輪郭にはラスタライズの差があります。全ての合成・印刷色を検証する資料ではありません。',
'page-geometry': '4ページ合計16個のマーカーを生成元の座標と照合済みです。',
'unembedded-font': 'DocViewはLiberation、PopplerはNimbusへ代替します。書体の形状差と文字欠落を区別してください。',
'icc-linear-rgb': '自作D65・sRGB原色・gamma 1のmatrix/TRCプロファイル。ICC LUT、校正済みCMYK、実モニターの発色は対象外です。',
}
extended_titles = {'japanese-layout': '埋込日本語・文字配置', 'graphics-compositing': '画像・透明・クリップ',
'page-geometry': 'ページ寸法・回転・切り抜き', 'unembedded-font': '非埋込Standard-14',
'icc-linear-rgb': 'ICC入力色(144 DPI)'}
for doc in comparisons['pdf-extended']['documents']:
stem = Path(doc['file']).stem
base = RESULTS / 'pdf-extended' / stem
for page in doc['pages']:
n = page['page']
checks = extended_checks[stem]
if stem == 'japanese-layout':
checks = (['「日本語の文字配置と埋込みフォント」が読める。',
'横組の本文と明示配置されたルビが読め、文字が重ならない。'] if n == 1 else
['縦列「日本語縦書き検証」「天地左右回転配置」の順序と配置を確認する。',
'0°・90°・180°・270°に回転した文字がそれぞれ表示される。'])
add(f'extended-{stem}-{n}', '追加PDF', f'{extended_titles[stem]}:{n}ページ',
TESTS / 'fixtures/pdf/extended' / doc['file'], doc['fixture_sha256'],
base / f'pdfium/page-{n}.png', base / f'poppler/page-{n}.png', base / f'difference-{n}.png',
page['expected_size_px'], checks, extended_notes[stem],
page['mean_absolute_channel_error'])
fonts = comparisons['pdf-fonts']
for page in fonts['pages']:
n = page['page']
add(f'japanese-unembedded-{n}', '日本語フォント', '非埋込CID:' + ('横書き' if n == 1 else '縦書き'),
TESTS / 'fixtures/pdf/fonts/unembedded-japanese.pdf', fonts['fixture_sha256'],
RESULTS / f'pdf-fonts/worker/page-{n}.png', RESULTS / f'pdf-fonts/poppler-{n}.png',
RESULTS / f'pdf-fonts/difference-{n}.png', page['size'],
['「日本語の文書を表示します。」「漢字とひらがな、カタカナ。」が読める。',
'「「縦書き」の句読点を確認。」の文字と句読点の位置、横組/縦組の順序を確認する。'],
'UniJIS-UCS2-H/V。DocViewはIPAexMincho / BIZ UDPGothicを代替に選択し、Popplerとは書体が異なります。'
'元のHeisei書体そのものの再現、全CMapやWindowsの互換性を示すものではありません。',
page['mean_absolute_channel_error'])
borders = comparisons['link-borders']
border_manifest = TESTS / 'fixtures/pdf/link-borders/manifest.json'
inputs.append(asset(border_manifest))
border_source = next(item for item in json.loads(border_manifest.read_text())['fixtures']
if item['file'] == 'styles.pdf')
if border_source['sha256'] != borders['fixture_sha256']:
raise ValueError('Link border manifest identity mismatch')
for page in borders['pages']:
n = page['page']
add(f'link-borders-{n}', 'リンク注釈', f'リンク枠:回転{page["rotation"]}°',
TESTS / 'fixtures/pdf/link-borders/styles.pdf', borders['fixture_sha256'],
RESULTS / f'link-borders/worker/page-{n}.png', RESULTS / f'link-borders/poppler-{n}.png',
RESULTS / f'link-borders/difference-{n}.png', page['whole_page']['size'],
['実線・破線の間隔・下線・BS優先・丸角・色・半透明・bevel/insetを下の位置表に沿って確認する。',
'zero-width、transparent、hidden、no-view、empty-appearance-preservedの枠は見えない。',
'appearance-preservedは緑の保存外観だけが見え、余分な赤枠が追加されない。'],
'Popplerの保存外観なしリンク補完はgray/CMYK、丸角、bevel/inset、alphaに制約があり、'
'保存AP/空APにも余分な枠を描きます。これらは右画像を正解にせず左の期待値を確認してください。'
'NoZoom/NoRotateの試験は別記録です。', page['whole_page']['mean_absolute_channel_error'])
regions = []
references = {item['name']: item['reference_supported'] for item in page['annotations']}
for annotation in border_source['annotations']:
x1, y1, x2, y2 = annotation['rect']
points = []
# Fixture CropBox [20,30,530,450], 144 DPI. Convert PDF bottom-left
# coordinates to the top-left of the actual rotated raster.
for x in (x1, x2):
for y in (y1, y2):
px, py = (x - 20) * 2, (450 - y) * 2
points.append([(px, py), (840 - py, px), (1020 - px, 840 - py),
(py, 1020 - px)][page['rotation'] // 90])
rect = [min(p[0] for p in points), min(p[1] for p in points),
max(p[0] for p in points), max(p[1] for p in points)]
name = annotation['name']
expectation = '表示される' if annotation['has_border'] else '表示されない'
if name == 'appearance-preserved':
expectation = '緑の保存外観だけが表示される(赤い追加枠なし)'
if name == 'empty-appearance-preserved':
expectation = '空の保存外観を保つ(追加枠なし)'
regions.append({'name': name, 'rect': rect, 'expectation': expectation,
'referenceSupported': references[name]})
cases[-1]['regions'] = regions
icc = comparisons['pdf-icc']
if icc['fixture_sha256_before'] != icc['fixture_sha256_after'] or not icc['all_source_assertions_pass']:
raise ValueError('ICC source assertions are not current/passing')
for scale in icc['scales']:
value = scale['scale']
base = RESULTS / 'pdf-icc' / ('scale-' + str(value).replace('.', '_'))
add(f'icc-scale-{value}', 'ICCと倍率', f'ICC:{value:g}倍({scale["dpi"]:g} DPI)',
TESTS / 'fixtures/pdf/extended/icc-linear-rgb.pdf', icc['fixture_sha256_before'],
base / 'pdfium/page-1.png', base / 'poppler.png', base / 'difference.png', scale['expected_size'],
extended_checks['icc-linear-rgb'] + ['この倍率の18色probeは理論値から各チャンネル差3以内。'],
extended_notes['icc-linear-rgb'], scale['mean_absolute_channel_error'])
collection = comparisons['font-collection']
for pair in collection['comparisons']:
name = pair['case']
runs = [r for r in collection['runs'] if r['case'] == name]
if len(runs) != 2 or runs[0]['source_sha256'] != runs[1]['source_sha256']:
raise ValueError('TTC source identity mismatch')
source = (RESULTS / 'font-collection/regular-bold.pdf' if name == 'paired-styles'
else TESTS / 'fixtures/pdf/extended/unembedded-font.pdf')
left = RESULTS / f'font-collection/standalone/{name}/page-1.png'
right = RESULTS / f'font-collection/collection/{name}/page-1.png'
asset(left, pair['standalone_png_sha256'])
asset(right, pair['collection_png_sha256'])
add(f'ttc-{name}', 'TTC/TTF(同一描画器)', 'TTCとTTF:' + ('通常/太字' if name == 'paired-styles' else 'Standard-14'),
source, runs[0]['source_sha256'], left, right, None, pair['dimensions'],
['左右が同じ画像で、TTCのface選択が個別TTFと一致する。'] +
(['同じ文「Collection face 0123456789 AVWA」の通常と太字が区別できる。'] if name == 'paired-styles' else []),
'同じPDFiumでフォントの格納形式を変えた比較です。独立レンダラーによる正解画像ではありません。'
'Latinの静的TrueTypeを対象とし、CJK/CFF/Windowsはこの比較に含みません。試験用フォント本体は回収済みです。',
pair['mean_absolute_channel_error'], independent=False)
if len(cases) != 21 or len({c['id'] for c in cases}) != 21:
raise ValueError('Expected exactly 21 distinct review cases')
template_path = TESTS / 'render_review/index.template.html'
template = template_path.read_text()
if template.count('__REVIEW_DATA__') != 1:
raise ValueError('Review template must contain one data placeholder')
data = {'schemaVersion': 1, 'workerSha256': worker, 'approvalStatus': 'pending-human-review',
'templateSha256': sha(template_path), 'generatorSha256': sha(Path(__file__)),
'comparisonInputs': inputs, 'cases': cases}
canonical = json.dumps(data, ensure_ascii=False, sort_keys=True, separators=(',', ':')).encode()
data['bundleSha256'] = hashlib.sha256(canonical).hexdigest()
serialized = json.dumps(data, ensure_ascii=False, indent=2) + '\n'
escaped = serialized.replace('&', '\\u0026').replace('<', '\\u003c').replace('>', '\\u003e')
OUTPUT.mkdir(parents=True, exist_ok=True)
(OUTPUT / 'manifest.json').write_text(serialized)
(OUTPUT / 'index.html').write_text(template.replace('__REVIEW_DATA__', escaped))
print(f'Generated {len(cases)} pairs: {data["bundleSha256"]}; approval remains pending.')
if __name__ == '__main__':
main()