#!/usr/bin/env python3 """Record bounded local provenance; never download, build, or execute DocView. This is an evidence ledger, not a license-compliance assessment. Package cache metadata is matched to the bundled inventory, not assumed to authenticate the installed files. Raw packager/build-directory/host inventory fields are omitted. Python 3.14+ supplies the streaming zstd tar reader used for Arch cache files. """ import argparse import hashlib import json from pathlib import Path, PurePosixPath import re import subprocess import tarfile import tempfile from package_linux_development import MANIFEST, ROOT, verify_and_extract from collect_linux_dependencies import qt_embedded_notices DEPENDENCIES = 'share/doc/docview/third-party/dependency-manifest.json' PRIMARY_CACHE_HASHES = {'qt6-base', 'qt6-declarative', 'qt6-webengine', 'tomlplusplus'} PKG_FIELDS = {'pkgname', 'pkgbase', 'pkgver', 'arch', 'license', 'builddate'} BUILD_FIELDS = {'format', 'pkgname', 'pkgbase', 'pkgver', 'pkgarch', 'pkgbuild_sha256sum', 'builddate', 'buildtool', 'buildtoolver'} MULTI_FIELDS = {'pkgname', 'license'} MAX_METADATA = 1024 * 1024 MAX_PREFIX = 8 * 1024 * 1024 MAX_ARCHIVE_HASH = 256 * 1024 * 1024 def digest(data): return hashlib.sha256(data).hexdigest() def file_digest(path): result = hashlib.sha256() with path.open('rb') as source: for chunk in iter(lambda: source.read(1024 * 1024), b''): result.update(chunk) return result.hexdigest() def canonical(value): return (json.dumps(value, ensure_ascii=False, sort_keys=True, indent=2) + '\n').encode() def bounded_bytes(path, maximum=MAX_METADATA): with path.open('rb') as source: data = source.read(maximum + 1) if len(data) > maximum: raise ValueError('Input exceeds its finite metadata bound') return data def parse_metadata(data, fields): """Discard sensitive fields before validating/exporting the allowed subset.""" if len(data) > MAX_METADATA or b'\0' in data: raise ValueError('Invalid package metadata size or encoding') result = {} for line in data.decode('utf-8', errors='strict').splitlines(): if not line or line.startswith('#'): continue key, separator, value = line.partition(' = ') if not separator: raise ValueError('Malformed package metadata record') if key not in fields: continue # The selected identity/license/build-tool fields never need paths, # email addresses, control characters, or freeform descriptions. if not value or len(value) > 512 or not re.fullmatch(r'[A-Za-z0-9_:.,+() /&*~=-]+', value): raise ValueError('Invalid selected package metadata value') if '/' in value or '@' in value: raise ValueError('Selected package metadata contains a path or address') if key in result and key not in MULTI_FIELDS: raise ValueError('Duplicate scalar package metadata') result.setdefault(key, []).append(value) return {key: sorted(values) if key in MULTI_FIELDS else values[0] for key, values in sorted(result.items())} def cache_metadata(path, expected): """Read only the bounded metadata prefix; do not unpack package payloads.""" raw = {} with tarfile.open(path, mode='r|zst') as archive: for index, member in enumerate(archive): if index >= 32 or member.offset_data + member.size > MAX_PREFIX: raise ValueError('Package metadata is outside the bounded prefix') if member.name in ('.PKGINFO', '.BUILDINFO'): if member.name in raw or not member.isfile() or member.size > MAX_METADATA: raise ValueError('Invalid package metadata member') raw[member.name] = archive.extractfile(member).read(MAX_METADATA + 1) if len(raw) == 2: break if set(raw) != {'.PKGINFO', '.BUILDINFO'}: raise ValueError('Package cache lacks required metadata') package = parse_metadata(raw['.PKGINFO'], PKG_FIELDS) build = parse_metadata(raw['.BUILDINFO'], BUILD_FIELDS) name, version, arch, base = expected for fields, arch_key in ((package, 'arch'), (build, 'pkgarch')): if (name not in fields.get('pkgname', []) or fields.get('pkgver') != version or fields.get(arch_key) != arch or fields.get('pkgbase', name) != base): raise ValueError('Cached package identity does not match the final inventory') if (build.get('format') != '2' or not re.fullmatch('[0-9a-f]{64}', build.get('pkgbuild_sha256sum', ''))): raise ValueError('Missing supported build format or PKGBUILD digest') return {'status': 'identity-matched-metadata', 'file': path.name, 'compressedBytes': path.stat().st_size, 'packageInfo': {'rawSha256': digest(raw['.PKGINFO']), 'selectedFields': package}, 'buildInfo': {'rawSha256': digest(raw['.BUILDINFO']), 'selectedFields': build}, 'signatureVerification': 'not-performed', 'installedFileToCachedPayloadComparison': 'not-performed', 'recipeCommit': None, 'recipeCommitStatus': 'PKGBUILD hash observed; Git revision not resolved by this collector; ' 'see separate source-correspondence records'} def cache_record(cache, name, component): version, arch = component['version'], component['architecture'] base = component['source']['packageBase'] if not all(re.fullmatch('[A-Za-z0-9_.+:~-]+', v) for v in (name, version, arch, base)): raise ValueError('Unsafe component identity in dependency manifest') path = cache / f'{name}-{version}-{arch}.pkg.tar.zst' if not path.is_file(): return {'status': 'not-present-in-selected-cache'} record = cache_metadata(path, (name, version, arch, base)) if name in PRIMARY_CACHE_HASHES: if path.stat().st_size > MAX_ARCHIVE_HASH: raise ValueError('Selected package exceeds bounded archive hashing limit') record['archiveSha256'] = file_digest(path) else: record['archiveHashStatus'] = 'not-computed; raw metadata digests recorded' return record def source_evidence(path, label, needles): data = bounded_bytes(path) lines = data.decode('utf-8').splitlines() return {'path': label, 'sha256': digest(data), 'selectedLines': [{'line': i, 'text': line.strip()} for i, line in enumerate(lines, 1) if any(needle in line for needle in needles)]} def inspect_elf(tool, arguments, executable, maximum=32 * 1024 * 1024): # readelf/nm inspect bytes. Neither the input executable nor ldd is run. with tempfile.TemporaryFile() as output: result = subprocess.run([tool, *arguments, str(executable)], stdout=output, stderr=subprocess.DEVNULL, timeout=20, check=False) if result.returncode or output.tell() > maximum: raise ValueError('Bounded ELF inspection failed') output.seek(0) return output.read(maximum + 1).decode('utf-8', errors='strict') def license_evidence(payload, row, bundled=False): relative = row['path'] if bundled else 'share/doc/docview/third-party/' + row['path'] path = PurePosixPath(relative) if path.is_absolute() or '..' in path.parts or str(path) != relative: raise ValueError('Unsafe license evidence path') data = bounded_bytes(payload / relative, 8 * MAX_METADATA) if len(data) != row['size'] or digest(data) != row['sha256']: raise ValueError('License text disagrees with packaged inventory') return {'payloadPath': relative, 'size': len(data), 'sha256': digest(data)} def qt_embedded_notice_record(payload, manifest, licenses, root): component = manifest['components']['qt6-webengine'] relative = 'licenses/qt6-webengine/embedded-resource-notices' full_relative = 'share/doc/docview/third-party/' + relative directory = payload / full_relative inventory = [row for row in licenses if row['payloadPath'].startswith(full_relative + '/')] supplied = component.get('embeddedResourceNotices') if supplied is None: if inventory or directory.exists(): raise ValueError('Qt embedded notice payload lacks manifest metadata') return {'status': 'not-in-this-package', 'completeChromiumNotices': False, 'scope': 'Legacy package; no embedded Qt resource notice claim'} original_path, metadata = qt_embedded_notices( manifest['host']['qtVersion'], component['version'], manifest['systemFiles'], root / 'resources/licenses/qt-embedded-notices') original = bounded_bytes(original_path, 65536) packaged = bounded_bytes(directory / 'sources.json', 65536) if original != packaged: raise ValueError('Qt embedded notice metadata differs from repository original') names = {row['name'] for row in metadata['files']} if {path.name for path in directory.iterdir()} != names | {'sources.json'}: raise ValueError('Unexpected Qt embedded notice payload contents') if component['source']['status'] != 'not-collected': raise ValueError('Partial Qt notices cannot claim complete source collection') evidence, expected_rows = [], [] for row in metadata['files']: data = bounded_bytes(directory / row['name']) source = bounded_bytes(original_path.parent / row['name']) if data != source or len(data) != row['size'] or digest(data) != row['sha256']: raise ValueError('Qt embedded notice differs from original or metadata') evidence.append({'payloadPath': full_relative + '/' + row['name'], 'size': len(data), 'sha256': digest(data)}) expected_rows.append({'path': relative + '/' + row['name'], 'size': row['size'], 'sha256': row['sha256'], 'origin': 'reviewed-installed-DataPack-resource-comment', 'sourceResources': row['sourceResources'], 'collectionScope': metadata['scope']}) if sorted(inventory, key=lambda row: row['payloadPath']) != sorted(evidence, key=lambda row: row['payloadPath']): raise ValueError('Qt embedded notice license inventory differs from metadata') recorded_rows = [row for row in component['licenseTexts'] if row['path'].startswith(relative + '/')] if sorted(recorded_rows, key=lambda row: row['path']) != sorted(expected_rows, key=lambda row: row['path']): raise ValueError('Qt embedded notice resource provenance differs from metadata') expected = {'status': 'collected-reviewed-resource-subset', 'metadataPath': relative + '/sources.json', 'metadataSha256': digest(original), 'noticeCount': len(metadata['files']), 'sourceResourceCount': sum(len(row['sourceResources']) for row in metadata['files']), 'runtimeDistribution': 'system-not-bundled', 'completeChromiumNotices': False, 'scope': metadata['scope']} if supplied != expected: raise ValueError('Qt embedded notice manifest metadata mismatch') return {'status': 'repository-original-and-package-matched', 'metadata': {'payloadPath': full_relative + '/sources.json', 'sha256': digest(original)}, 'repositoryMetadata': {'path': 'resources/licenses/qt-embedded-notices/sources.json', 'sha256': digest(original)}, 'noticeCount': expected['noticeCount'], 'sourceResourceCount': expected['sourceResourceCount'], 'completeChromiumNotices': False, 'runtimeDistribution': 'system-not-bundled', 'sourceCollectionStatus': 'not-collected', 'dataPacks': metadata['dataPacks'], 'files': expected_rows, 'scope': metadata['scope'], 'scopeLimit': 'Repository originals and packaged notice bytes matched to the recorded system DataPack inventory; ' 'host DataPacks were not rehashed here. No complete Chromium attribution, source collection, or legal assessment.'} def pdfium_supplemental_record(payload, component, licenses, lock, library_hash, root): relative = 'share/doc/docview/pdfium/supplemental' inventory = [row for row in licenses if row['payloadPath'].startswith(relative + '/')] supplied = component.get('supplementalNotices') directory = payload / relative if supplied is None: if inventory or directory.exists(): raise ValueError('PDFium supplemental payload lacks manifest metadata') return {'status': 'not-in-this-package', 'scope': 'Legacy provider-only notices; no supplemental notice claim'} original_root = root / 'resources/licenses/pdfium-supplemental' original = bounded_bytes(original_root / 'sources.json', 65536) packaged = bounded_bytes(directory / 'sources.json', 65536) if original != packaged: raise ValueError('PDFium supplemental metadata differs from repository original') metadata = json.loads(original) if (type(metadata.get('schemaVersion')) is not int or metadata['schemaVersion'] != 1 or metadata.get('pdfiumVersion') != lock['version'] or metadata.get('pdfiumUpstreamCommit') != lock['upstreamCommit'] or metadata.get('pdfiumLibrarySha256') != library_hash): raise ValueError('PDFium supplemental source pin or library hash mismatch') rows = metadata.get('files') names = {'libcxx-LICENSE.txt', 'libcxxabi-LICENSE.txt'} if (not isinstance(rows, list) or len(rows) != 2 or not all(isinstance(row, dict) for row in rows) or {row.get('name') for row in rows} != names): raise ValueError('Unexpected PDFium supplemental notice set') if (not re.fullmatch('[0-9a-f]{40}', metadata.get('providerRecipeCommit', '')) or {path.name for path in directory.iterdir()} != names | {'sources.json'}): raise ValueError('Invalid PDFium supplemental recipe or payload contents') files, evidence = [], [] for row in rows: if (type(row.get('size')) is not int or not 0 < row['size'] <= MAX_METADATA or not re.fullmatch('[0-9a-f]{64}', row.get('sha256', '')) or not re.fullmatch('[0-9a-f]{40}', row.get('sourceRevision', '')) or not isinstance(row.get('sourceUrl'), str) or not row['sourceUrl'].startswith('https://chromium.googlesource.com/') or '/+/' + row['sourceRevision'] + '/' not in row['sourceUrl']): raise ValueError('Invalid PDFium supplemental fixed-source notice metadata') source = bounded_bytes(original_root / row['name']) data = bounded_bytes(directory / row['name']) if (source != data or len(source) != row['size'] or digest(source) != row['sha256']): raise ValueError('PDFium supplemental notice differs from original or metadata') path = relative + '/' + row['name'] files.append({**row, 'path': path}) evidence.append({'payloadPath': path, 'size': len(data), 'sha256': digest(data)}) if sorted(inventory, key=lambda row: row['payloadPath']) != sorted(evidence, key=lambda row: row['payloadPath']): raise ValueError('PDFium supplemental license inventory differs from metadata') expected = {'metadataPath': relative + '/sources.json', 'metadataSha256': digest(original), 'scope': metadata['scope'], 'providerRecipeCommit': metadata['providerRecipeCommit'], 'pdfiumLibrarySha256': library_hash, 'files': files} if supplied != expected: raise ValueError('PDFium supplemental manifest metadata mismatch') return {'status': 'repository-original-and-package-matched', **expected, 'repositoryMetadata': {'path': 'resources/licenses/pdfium-supplemental/sources.json', 'sha256': digest(original)}, 'scopeLimit': 'Fixed-source notice bytes and recorded pins matched; ' 'no new source download, build reproduction or legal assessment'} def pdfium_record(payload, manifest, pdfium_root, root): component = manifest['components']['PDFium-independent-worker'] lock_data = bounded_bytes(root / 'cmake/pdfium.lock.json') lock = json.loads(lock_data) version_data = bounded_bytes(pdfium_root / 'VERSION') values = dict(re.findall(r'^(MAJOR|MINOR|BUILD|PATCH)=(\d+)$', version_data.decode(), re.M)) if set(values) != {'MAJOR', 'MINOR', 'BUILD', 'PATCH'}: raise ValueError('Invalid PDFium VERSION metadata') version = '.'.join(values[key] for key in ('MAJOR', 'MINOR', 'BUILD', 'PATCH')) args_data = bounded_bytes(pdfium_root / 'args.gn') args = {} for line in args_data.decode().splitlines(): match = re.fullmatch(r'([a-z_][a-z_0-9]*)\s*=\s*(true|false|"[A-Za-z0-9_-]+")', line.strip()) if not match or match[1] in args: raise ValueError('Unsupported or duplicate PDFium build argument') args[match[1]] = json.loads(match[2]) if (version != component['version'] or version != lock['version'] or lock['upstreamCommit'] != component['source']['upstreamCommit'] or lock['linuxX64Sha256'] != component['source']['archiveSha256'] or any(args.get(key) != value for key, value in lock['buildOptions'].items())): raise ValueError('Local PDFium build metadata disagrees with final inventory/lock') packaged_library = payload / component['path'] library_hash = file_digest(packaged_library) if (library_hash != component['sha256'] or library_hash != file_digest(pdfium_root / 'lib/libpdfium.so')): raise ValueError('Local PDFium library differs from packaged library') licenses = [license_evidence(payload, row, True) for row in component['licenseTexts']] if len({row['payloadPath'] for row in licenses}) != len(licenses): raise ValueError('Duplicate PDFium license inventory entry') for row in licenses: if not row['payloadPath'].startswith('share/doc/docview/pdfium/'): raise ValueError('Unexpected PDFium license prefix') suffix = row['payloadPath'].removeprefix('share/doc/docview/pdfium/') if suffix.startswith('supplemental/'): continue if file_digest(pdfium_root / suffix) != row['sha256']: raise ValueError('Local PDFium license differs from package') supplements = pdfium_supplemental_record(payload, component, licenses, lock, library_hash, root) return {'version': version, 'library': {'payloadPath': component['path'], 'sha256': library_hash, 'size': packaged_library.stat().st_size}, 'localLibraryMatchesPayload': True, 'licenseTexts': licenses, 'supplementalNotices': supplements, 'providerLicense': 'MIT (top-level LICENSE; PDFium upstream license is separate)', 'pdfiumUpstreamLicense': 'BSD-3-Clause text in licenses/pdfium.txt; see full file', 'buildMetadata': { 'VERSION': {'sha256': digest(version_data), 'values': values}, 'args.gn': {'sha256': digest(args_data), 'values': args}, 'lock': {'path': 'cmake/pdfium.lock.json', 'sha256': digest(lock_data)}}, 'pinnedProviderArchive': {'url': lock['linuxX64Archive'], 'sha256FromLock': lock['linuxX64Sha256'], 'archiveReverifiedThisRun': False}, 'upstreamCommit': lock['upstreamCommit'], 'upstream': lock['upstream'], 'binaryProvider': lock['binaryProvider'], 'remaining': ['Provider recipe/patch correspondence is not independently verified by this collector; ' 'see tests/results/source-correspondence/pdfium records', 'Transitive source correspondence is not independently verified by this collector; ' 'see separate source-correspondence records', 'Completeness of corresponding source and applicable obligations is not assessed here']} def make_record(archive, cache, pdfium_root, root=ROOT): with tempfile.TemporaryDirectory(prefix='docview-provenance-') as temporary: payload = verify_and_extract(archive, Path(temporary) / 'verified') manifest_bytes = bounded_bytes(payload / DEPENDENCIES, 16 * MAX_METADATA) manifest = json.loads(manifest_bytes) package_manifest_bytes = bounded_bytes(payload / MANIFEST, 8 * MAX_METADATA) package_manifest = json.loads(package_manifest_bytes) payload_rows = package_manifest['files'] runtime_rows = [row for row in payload_rows if row['path'].startswith(('bin/', 'lib/'))] if {row['path'] for row in runtime_rows} != { 'bin/docview', 'bin/docview-pdf-worker', 'bin/docview-archive-worker', 'lib/libpdfium.so'}: raise ValueError('Unexpected runtime payload; distribution scope must be reviewed') components = {} for name, item in sorted(manifest['components'].items()): if name == 'PDFium-independent-worker': continue if item['distribution'] != 'system-not-bundled': raise ValueError('Unreviewed dependency distribution category') files = [row for row in manifest['systemFiles'] if row['package'] == name] components[name] = { 'version': item['version'], 'architecture': item['architecture'], 'runtimeBinaryDistribution': 'system-only; not in this tar', 'headerCodePresence': ('defined toml symbols observed in packaged DocView' if name == 'tomlplusplus' else 'not-audited'), 'usageFromManifest': item.get('usage', []), 'licenseLabelsFromPackage': item['licenseLabelsFromPackage'], 'licenseTexts': [license_evidence(payload, row) for row in item['licenseTexts']], 'sourceCollectionStatus': item['source']['status'], 'sourceCollectionStatusScope': 'Status copied from this packaged inventory; ' 'separate source-correspondence records may contain later collection evidence', 'obligationAssessment': 'not-performed', 'recipeRepository': item['source']['recipeRepository'], 'recordedSystemFiles': {'count': len(files), 'canonicalRowsSha256': digest(canonical(files)), 'evidence': 'packaged dependency-manifest.json#/systemFiles', 'hostFilesRehashedThisRun': False}, 'cachedPackage': cache_record(cache, name, item)} if name == 'qt6-webengine': components[name]['embeddedResourceNotices'] = qt_embedded_notice_record( payload, manifest, components[name]['licenseTexts'], root) dynamic, embedded = [], {} for relative in ('bin/docview', 'bin/docview-pdf-worker', 'bin/docview-archive-worker'): text = inspect_elf('readelf', ['--dynamic', '--wide'], payload / relative) needed = sorted(re.findall(r'\(NEEDED\).*?\[([^\]]+)\]', text)) if not needed or not any(name.startswith('libQt6') for name in needed): raise ValueError('Expected dynamic Qt dependency was not found') dynamic.append({'payloadPath': relative, 'directNeeded': needed}) text = inspect_elf('nm', ['--defined-only', '--demangle'], payload / 'bin/docview') symbols = sorted({line.split(' ', 2)[-1] for line in text.splitlines() if re.match(r'^[0-9a-fA-F]+ [A-Za-z] toml::', line)}) if not symbols: raise ValueError('No defined toml symbols; header-code inference needs review') embedded = {'component': 'tomlplusplus', 'payloadPath': 'bin/docview', 'definedSymbolCount': len(symbols), 'definedSymbols': symbols, 'interpretation': 'Header-derived code is present in this executable; ' 'the toml++ shared-library binary is still system-only.', 'sourceEvidence': source_evidence(root / 'src/core/config.cpp', 'src/core/config.cpp', ['#include '])} qt_files = [row for row in manifest['systemFiles'] if row['package'].startswith('qt6-') and re.fullmatch(r'/usr/lib/libQt6[^/]+\.so(?:\.[0-9]+)+', row['path'])] pdfium = pdfium_record(payload, manifest, pdfium_root, root) return {'schemaVersion': 1, 'scope': 'Local final Arch development package evidence', 'legalComplianceVerdict': 'not-assessed', 'docviewLicense': 'unspecified', 'sourceCollectionIsNotObligationAssessment': True, 'archive': {'file': archive.name, 'sha256': file_digest(archive), 'compressedBytes': archive.stat().st_size, 'payloadFileCount': len(payload_rows) + 1, 'payloadBytes': sum(row['size'] for row in payload_rows) + len(package_manifest_bytes), 'payloadIntegrity': 'all members verified against packaged payload manifest', 'manifestExcludesOwnDigest': True, 'packageManifestSha256': digest(package_manifest_bytes), 'dependencyManifestSha256': digest(manifest_bytes), 'runtimeFiles': runtime_rows}, 'tools': {name: inspect_elf(name, ['--version'], Path('/dev/null'), MAX_METADATA) .splitlines()[0] for name in ('nm', 'readelf')}, 'pdfium': pdfium, 'systemComponents': components, 'dynamicLinkEvidence': {'method': 'readelf DT_NEEDED on verified packaged executables', 'executables': dynamic, 'qtLibraryRowsFromPackagedInventory': qt_files, 'qtAndWebEngineRuntimeBinariesInPayload': False, 'scope': 'Direct ELF links and recorded system resolutions; no claim that ' 'all Qt header-derived machine code is absent'}, 'embeddedHeaderEvidence': embedded, 'privacy': {'omitted': ['home paths', 'user names', 'packager identities and emails', 'build directories', 'complete builder installed-package inventory'], 'rawCacheMetadataCopied': False}, 'limits': ['Cache metadata is identity-matched, not signature-authenticated', 'Cache metadata is read only from its bounded prefix; the rest of each cache tar is not validated', 'Only four primary cache archives receive a whole-archive SHA-256', 'PKGBUILD SHA-256 is evidence of a recipe digest, not a recovered recipe or Git commit', 'Recorded installed file hashes are from the final package manifest, not remeasured here', 'This collector does not independently collect or verify complete source trees, recipe patches, ' 'or WebEngine build-specific full Chromium notices; see separate source-correspondence records', 'This is not a reproducible-build, clean-OS, target-OS, or license-compliance result']} def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('--archive', type=Path, default=ROOT / 'tests/results/linux-development-package/final/docview-0.1.0-arch-x86_64-development.tar.gz') parser.add_argument('--package-cache', type=Path, default=Path('/var/cache/pacman/pkg')) parser.add_argument('--pdfium-root', type=Path, default=ROOT / '.deps/pdfium') parser.add_argument('--output', type=Path, default=ROOT / 'tests/results/dependency-provenance/local-final') args = parser.parse_args() args.output.mkdir(parents=True, exist_ok=True) target = args.output / 'provenance.json' if target.exists(): raise ValueError('Output already exists; choose a new evidence directory') record = make_record(args.archive, args.package_cache, args.pdfium_root) data = canonical(record) target.write_bytes(data) matched = sum(c['cachedPackage']['status'] == 'identity-matched-metadata' for c in record['systemComponents'].values()) print(json.dumps({'recordSha256': digest(data), 'payloadFileCount': record['archive']['payloadFileCount'], 'cacheMetadataMatched': matched, 'systemComponents': len(record['systemComponents'])})) if __name__ == '__main__': main()