initial commit

This commit is contained in:
2026-09-21 13:41:40 +09:00
commit 855c7328df
411 changed files with 85352 additions and 0 deletions
+47
View File
@@ -0,0 +1,47 @@
# 固定版ソースアーカイブの取得・照合
Qt 6.11.2とtoml++ 3.4.0の公開ソースを取得し、固定した公開hashと照合する。ダウンロードしたビルドスクリプトは実行しない。Qtでは通常ファイルを専用の新規treeへ保存し、path、件数、単体・総容量に上限を設ける。リンクを辿らず、元アーカイブのpath・mode・全通常ファイルのSHA-256をJSONLへ記録する。ファイルの実行属性は設定しないため、抽出treeをそのままビルド可能なcheckoutとは呼ばない。
```sh
python3 tests/source_archives/fetch.py \
--output tests/results/source-archives/new-download \
--cache .deps/new-source-archives
python3 tests/source_archives/inspect_qt.py \
--archive .deps/new-source-archives/qt-everywhere-src-6.11.2.tar.xz \
--tree .deps/new-qt-regular-files \
--output tests/results/source-archives/new-qt-inspection
python3 tests/source_archives/verify_toml.py \
--output tests/results/source-archives/new-toml-check
python3 tests/source_archives/check_arch_patches.py \
--output tests/results/source-archives/new-patch-check
```
各出力先は新規にする。`verify_toml.py`は既定の`.deps/source-archives`を、patch確認は保存済み`qt-inspection/report.json`のtreeを読む。既存の資料を更新する場合は、対応するpinと参照箇所を明示的に変更する。
Qtのpinは[公式mirror metadata](https://download.qt.io/archive/qt/6.11/6.11.2/single/qt-everywhere-src-6.11.2.tar.xz.mirrorlist)、toml++は保存済みの使用版Arch recipeによる。Qtのmirrorページには要求元IPが含まれることがあるため、全文は保存せず、応答hashと一致した公開checksumを記録する。
[今回の実行結果](../results/source-archives/README.md)。ソース取得、使用版との部分的な照合、実ビルド構成に対する告知の完全性、具体的配布条件の判定は区別する。
## PDFiumの固定Gitソース
保存済みDEPSの固定値からLinux / Windows x64 minimalのGit取得対象を選び、PDFium本体と27依存を取得する。3並列、fetch上限900秒、Git hookとcredential helper無効で、DEPSや取得ソースのコードは実行しない。Git objectを検査してから通常ファイル・リンクの全blobをアーカイブ化し、読み戻して照合する。アーカイブに保存したリンクはファイルシステムへ展開しない。
```sh
python3 tests/source_archives/collect_pdfium.py \
--output tests/results/source-archives/new-pdfium-git \
--cache .deps/new-pdfium-git
python3 tests/source_archives/check_pdfium.py \
--collection tests/results/source-archives/new-pdfium-git \
--output tests/results/source-archives/new-pdfium-check
python3 tests/source_archives/collect_pdfium_gitlink.py \
--output tests/results/source-archives/new-pdfium-gitlink \
--cache .deps/new-pdfium-gitlink
```
出力先とアーカイブのcacheは新規にする。rootのbare repositoryのみ、存在すれば最初の取得確認用cacheを再利用して固定commit・treeを再検査する。gitlink補足スクリプトは今回の既定`pdfium-git/report.json`と固定したFreeType参照を読み、未確認の追加参照があれば停止する。
checkスクリプトはproviderのパッチを出力先内の専用コピーへ適用する。Linux3件とWindows4件の差分、既存Linux配布物の全24公開ヘッダー・15告知を比較し、本体・ワーカーを変更しない。[PDFiumの実行結果と範囲](../results/source-archives/PDFIUM.md)
## Ubuntu用Qt SDKの告知
`collect_qt_sdk_notices.py --output <新規directory>`は、保存済みの公式Qt SDK archive、Ubuntu実行時台帳とSDK SBOM、検証済みQt sourceからChromiumの告知本文を抽出する。全archiveをストリームで検査し、SBOMの67ファイル・既存実行時台帳の83ファイル・告知129entryを照合する。元SBOMの参照ID不整合や本文のない項目、生成器のskipを保持し、網羅性の判定とは分ける。[実行結果](../results/source-archives/QT-SDK-NOTICES.md)
@@ -0,0 +1,86 @@
#!/usr/bin/env python3
"""Check recorded Arch patches on selected pristine release sources, without building."""
import argparse
import hashlib
import json
from pathlib import Path, PurePosixPath
import shutil
import subprocess
ROOT = Path(__file__).resolve().parents[2]
def sha(path):
return hashlib.sha256(path.read_bytes()).hexdigest()
def main():
p = argparse.ArgumentParser(description=__doc__)
p.add_argument('--output', type=Path, required=True)
a = p.parse_args(); output = a.output.resolve()
if not output.is_relative_to(ROOT):
p.error('Use an output directory inside the workspace')
inspection = ROOT / 'tests/results/source-archives/qt-inspection'
record = json.loads((inspection / 'report.json').read_text())
if not record['success'] or sha(inspection / 'files.jsonl') != record['inventorySha256']:
raise SystemExit('A verified source inventory is required')
inventory = {row['path']: row for line in (inspection / 'files.jsonl').open() if (row := json.loads(line))}
source = ROOT / record['sourceTree']
evidence = ROOT / 'tests/results/source-correspondence/arch'
patches = [
('qtbase', evidence / 'packages/qt6-base/qt6-base-cflags.patch'),
('qtbase', evidence / 'packages/qt6-base/qt6-base-nostrip.patch'),
('qtbase', evidence / 'upstream-metadata/qtbase-e80e3f0.patch'),
('qtdeclarative', evidence / 'upstream-metadata/qtdeclarative-2efb7c6.patch')]
output.mkdir(parents=True, exist_ok=False)
inputs, steps = {}, []
for module, patch in patches:
for line in patch.read_text().splitlines():
if not line.startswith('+++ b/'):
continue
name = line[len('+++ b/'):].split('\t', 1)[0]
path = PurePosixPath(name)
if path.is_absolute() or '..' in path.parts:
raise ValueError('Unsafe patch target')
relative = module + '/' + name
target = output / 'tree' / relative
target.parent.mkdir(parents=True, exist_ok=True)
if relative in inputs:
continue
before = None
if (source / relative).exists():
before = sha(source / relative)
if before != inventory[relative]['sha256']:
raise ValueError('Selected release file changed')
shutil.copyfile(source / relative, target)
inputs[relative] = before
success = True
for module, patch in patches:
# The two local Arch patches use GNU patch's default fuzz tolerance.
# Keep exact context for the separately recorded upstream changes.
fuzz = 2 if patch.name.startswith('qt6-base-') else 0
step = {'module': module, 'patch': str(patch.relative_to(ROOT)), 'patchSha256': sha(patch), 'fuzzLimit': fuzz}
for label, extra in [('dryRun', ['--dry-run']), ('apply', [])]:
command = ['patch', '--batch', '--forward', '--fuzz=' + str(fuzz), '-p1', '-d', str(output / 'tree' / module), '-i', str(patch), *extra]
run = subprocess.run(command, capture_output=True, text=True, timeout=20)
step[label] = {'exitCode': run.returncode, 'stdout': run.stdout, 'stderr': run.stderr}
if run.returncode:
success = False; break
steps.append(step)
if not success:
break
after = {name: sha(output / 'tree' / name) if (output / 'tree' / name).is_file() else None for name in inputs}
report = {'success': success, 'sourceArchiveSha256': record['archiveSha256'],
'beforeSha256': inputs, 'afterSha256': after, 'steps': steps,
'pristineFilesUnchanged': all(sha(source / name) == digest for name, digest in inputs.items() if digest),
'scope': 'Recorded Arch patches checked on a private selected-file copy with explicit tolerances; no upstream build scripts run and no resulting binaries rebuilt'}
report['installedMkspecComparisons'] = {
name: sha(output / 'tree/qtbase/mkspecs/common' / name) == sha(Path('/usr/lib/qt6/mkspecs/common') / name)
for name in ['g++-unix.conf', 'gcc-base.conf']}
(output / 'report.json').write_text(json.dumps(report, indent=2) + '\n')
print(json.dumps({'success': success, 'patches': len(steps), 'targetFiles': len(inputs), 'pristineFilesUnchanged': report['pristineFilesUnchanged']}))
return 0 if success else 1
if __name__ == '__main__':
raise SystemExit(main())
+176
View File
@@ -0,0 +1,176 @@
#!/usr/bin/env python3
"""Compare collected PDFium source, provider patches and installed public headers."""
import argparse
import hashlib
import importlib.util
import json
from pathlib import Path, PurePosixPath
import subprocess
from collect_pdfium import ROOT, EVIDENCE, sha, run, save
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--collection', type=Path, default=ROOT / 'tests/results/source-archives/pdfium-git')
parser.add_argument('--output', type=Path, required=True)
args = parser.parse_args()
output = args.output.resolve()
if not output.is_relative_to(ROOT):
parser.error('Use an output directory inside the workspace')
collection = args.collection.resolve()
report_path = collection / 'report.json'
collected = json.loads(report_path.read_text())
if not collected['success']:
raise ValueError('Complete successful selected-Git collection required')
repositories, inventories = {}, {}
for item in collected['repositories']:
if sha(ROOT / item['archive']) != item['archiveSha256'] or sha(ROOT / item['inventory']) != item['inventorySha256']:
raise ValueError('Source archive or inventory changed')
repositories[item['path']] = item
inventories[item['path']] = {e['path']: e for line in (ROOT / item['inventory']).open() if (e := json.loads(line))}
def blob(repository, path):
item = repositories[repository]
expected = inventories[repository][path]
if expected['mode'] not in {'100644', '100755'}:
raise ValueError('Expected a regular source file')
result = run(ROOT / item['repository'], ['cat-file', 'blob', expected['gitObject']])
result.check_returncode()
if hashlib.sha256(result.stdout).hexdigest() != expected['sha256']:
raise ValueError('Source blob differs from inventory')
return result.stdout
def locate(saved):
if saved.startswith('upstream/'):
return '.', saved.removeprefix('upstream/')
if saved.startswith('dependencies/'):
relative = saved.removeprefix('dependencies/')
key = next(k for k in sorted(repositories, key=len, reverse=True) if relative.startswith(k + '/'))
return key, relative[len(key) + 1:]
raise ValueError('Unexpected saved source')
output.mkdir(parents=True, exist_ok=False)
# Tie earlier Gitiles bytes to the full fetched snapshots, not just filenames.
correspondence = []
for folder in ('upstream', 'dependencies'):
for sidecar in sorted((EVIDENCE / folder).rglob('*.retrieval.json')):
retrieval = json.loads(sidecar.read_text())
if retrieval['transform'] != 'base64 decoded Gitiles response':
continue
saved = retrieval['file']
repository, path = locate(saved)
content = blob(repository, path)
digest = hashlib.sha256(content).hexdigest()
if digest != retrieval['sha256'] or content != (EVIDENCE / saved).read_bytes():
raise ValueError('Saved Gitiles source differs: ' + saved)
correspondence.append({'saved': saved, 'repository': repository, 'path': path,
'revision': repositories[repository]['revision'], 'sha256': digest})
provider = EVIDENCE / 'provider/recipe'
provider_report = json.loads((EVIDENCE / 'report.json').read_text())
recipe_hashes = {e['file']: e['sha256'] for e in provider_report['recipeFilesVerifiedAgainstGitTree']}
for name in ('steps/03-patch.sh', 'steps/07-stage.sh', 'steps/08-licenses.sh'):
if sha(provider / name) != recipe_hashes[name]:
raise ValueError('Provider recipe changed')
selections = [('patches/shared_library.patch', '.'), ('patches/public_headers.patch', '.'),
('patches/clang_rt.patch', 'build'), ('patches/win/build.patch', 'build')]
binaries = {name: sha(ROOT / 'build' / name) for name in ['docview', 'docview-pdf-worker', 'docview-archive-worker']}
platforms = {}
for platform, count in [('linux', 3), ('win', 4)]:
tree = output / platform / 'tree'
before = {}
# Include the entire public directory so the package header inventory is
# compared in both directions after applying the provider's text patch.
for path, entry in inventories['.'].items():
if path.startswith('public/') and entry['mode'] in {'100644', '100755'}:
target = tree / path
target.parent.mkdir(parents=True, exist_ok=True)
target.write_bytes(blob('.', path))
before[path] = entry['sha256']
steps = []
for name, repository in selections[:count]:
patch = provider / name
if sha(patch) != recipe_hashes[name]:
raise ValueError('Provider patch changed')
work = tree if repository == '.' else tree / repository
for line in patch.read_text().splitlines():
if not line.startswith('+++ b/'):
continue
path = line[len('+++ b/'):].split('\t')[0]
safe = PurePosixPath(path)
if safe.is_absolute() or '..' in safe.parts:
raise ValueError('Unsafe patch path')
target = work / path
target.parent.mkdir(parents=True, exist_ok=True)
full_path = str(target.relative_to(tree))
if full_path not in before:
target.write_bytes(blob(repository, path))
before[full_path] = sha(target)
step = {'patch': str(patch.relative_to(ROOT)), 'sha256': sha(patch), 'directory': repository, 'fuzzLimit': 2}
for phase, extra in [('dryRun', ['--dry-run']), ('apply', [])]:
applied = subprocess.run(['patch', '--batch', '--forward', '--verbose', '--fuzz=2', '-p1',
'-d', str(work), '-i', str(patch), *extra],
capture_output=True, text=True, timeout=20)
step[phase] = {'exitCode': applied.returncode, 'stdout': applied.stdout, 'stderr': applied.stderr}
if applied.returncode:
save(output / 'failure.json', step)
raise ValueError('Recorded provider patch failed: ' + name)
steps.append(step)
expected_headers = {str(p.relative_to(tree / 'public')): p for p in (tree / 'public').rglob('*.h')}
packaged_headers = {str(p.relative_to(ROOT / '.deps/pdfium/include')): p
for p in (ROOT / '.deps/pdfium/include').rglob('*.h')}
if expected_headers.keys() != packaged_headers.keys():
raise ValueError('Public header inventory differs')
header_checks = [{'path': name, 'sourceSha256': sha(path), 'packagedSha256': sha(packaged_headers[name]),
'match': path.read_bytes() == packaged_headers[name].read_bytes()}
for name, path in sorted(expected_headers.items())]
if not all(e['match'] for e in header_checks):
raise ValueError('Patched public headers differ')
# Provider GNU patch can retain an original file after an offset/fuzzy
# application. Keep the observed .orig file separate from API headers.
extras = []
for path in (ROOT / '.deps/pdfium/include').rglob('*'):
if not path.is_file() or path.suffix == '.h':
continue
relative = str(path.relative_to(ROOT / '.deps/pdfium/include'))
matches = relative.endswith('.orig') and path.read_bytes() == blob('.', 'public/' + relative[:-5])
extras.append({'path': relative, 'sha256': sha(path), 'matchesPristineSource': matches})
if not matches:
raise ValueError('Unexplained packaged extra header file')
platforms[platform] = {'steps': steps, 'beforeSha256': before,
'afterSha256': {p: sha(tree / p) for p in before},
'linuxPackagedHeaderComparisons': header_checks, 'extraPackagedFiles': extras,
'windowsResourceGenerated': False, 'compiled': False}
spec = importlib.util.spec_from_file_location('notice_correspondence', EVIDENCE / 'check_correspondence.py')
notices = importlib.util.module_from_spec(spec); spec.loader.exec_module(notices)
notice_checks = []
for packaged, (source, mode) in notices.MAPPING.items():
if source.startswith('provider/'):
content = (EVIDENCE / source).read_bytes()
if sha(EVIDENCE / source) != recipe_hashes['LICENSE']:
raise ValueError('Provider license changed')
else:
content = blob(*locate(source))
transformed = notices.transform(content, mode)
actual = ROOT / '.deps/pdfium' / packaged
match = transformed == actual.read_bytes()
notice_checks.append({'path': packaged, 'sha256': sha(actual), 'match': match, 'transform': mode})
if not match:
raise ValueError('Notice content differs')
after_binaries = {name: sha(ROOT / 'build' / name) for name in binaries}
if binaries != after_binaries:
raise ValueError('Production binaries changed')
report = {'success': True, 'collectionReportSha256': sha(report_path),
'checkerSha256': sha(Path(__file__)), 'gitilesSourceComparisons': correspondence,
'platforms': platforms, 'noticeComparisons': notice_checks,
'productionBinaries': binaries, 'productionBinariesUnchanged': True,
'scope': 'Source text, selected Linux/Windows provider patches and installed Linux public headers/notices only. '
'No build scripts, Windows compiler/resource generation or reproducible binary build executed.'}
save(output / 'report.json', report)
print(json.dumps({'success': True, 'gitilesComparisons': len(correspondence),
'publicHeaders': len(expected_headers), 'notices': len(notice_checks),
'patchApplications': sum(len(p['steps']) for p in platforms.values())}))
if __name__ == '__main__':
main()
+246
View File
@@ -0,0 +1,246 @@
#!/usr/bin/env python3
"""Collect fixed PDFium Git snapshots without executing checkout/build scripts."""
import argparse
from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import datetime, timezone
import gzip
import hashlib
import io
import json
import os
from pathlib import Path, PurePosixPath
import re
import subprocess
import tarfile
from urllib.parse import urlsplit
ROOT = Path(__file__).resolve().parents[2]
EVIDENCE = ROOT / 'tests/results/source-correspondence/pdfium'
PINS = EVIDENCE / 'dependency-pins.json'
ENV = dict(os.environ, GIT_CONFIG_GLOBAL='/dev/null', GIT_CONFIG_NOSYSTEM='1',
GIT_TERMINAL_PROMPT='0', GIT_CONFIG_COUNT='0', LC_ALL='C')
GIT = ['git', '-c', 'core.hooksPath=/dev/null', '-c', 'credential.helper=',
'-c', 'protocol.file.allow=never', '-c', 'transfer.fsckObjects=true',
'-c', 'fetch.fsckObjects=true', '-c', 'gc.auto=0']
def sha(path):
with path.open('rb') as stream:
return hashlib.file_digest(stream, 'sha256').hexdigest()
def save(path, value):
path.write_text(json.dumps(value, indent=2, ensure_ascii=False) + '\n')
def run(repo, arguments, timeout=120):
command = GIT + ['--git-dir=' + str(repo), *arguments]
return subprocess.run(command, env=ENV, capture_output=True, timeout=timeout)
def plan():
data = json.loads(PINS.read_text())
root = json.loads((EVIDENCE / 'upstream/commit.json').read_text().removeprefix(")]}'\n"))
if root['commit'] != data['rootRevision']:
raise ValueError('Root commit evidence differs')
selected = [{'path': '.', 'url': 'https://pdfium.googlesource.com/pdfium.git',
'revision': data['rootRevision'], 'expectedTree': root['tree'], 'condition': None}]
excluded, packages = [], []
# Explicit union of Linux x64 and Windows x64 minimal, V8/Rust/Skia-off
# checkouts. Refuse new predicates instead of executing any DEPS code.
conditions = {'checkout_libpng': True, 'checkout_win': True,
'checkout_testing_corpus': False, 'checkout_rust': False,
'checkout_android': False, 'checkout_v8': False, 'checkout_skia': False}
for name, value in data['root']['deps'].items():
if isinstance(value, str):
url, condition = value, None
elif 'url' in value:
url, condition = value['url'], value.get('condition')
else:
packages.append({'path': name, 'pin': value})
continue
if condition is not None and condition not in conditions:
raise ValueError('Unreviewed Git condition: ' + condition)
url, revision = url.rsplit('@', 1)
parsed = urlsplit(url)
if (parsed.scheme != 'https' or parsed.netloc not in {
'pdfium.googlesource.com', 'chromium.googlesource.com', 'skia.googlesource.com'}
or not re.fullmatch('[0-9a-f]{40}', revision)):
raise ValueError('Unexpected source URL or revision')
item = {'path': name, 'url': url, 'revision': revision, 'condition': condition}
(selected if condition is None or conditions[condition] else excluded).append(item)
for name, recursive in data['recursive'].items():
for subpath, value in recursive['deps'].items():
if isinstance(value, str) or 'url' in value:
raise ValueError('Unreviewed recursive Git source')
packages.append({'path': name + '/' + subpath, 'pin': value})
return {'selectedGit': selected, 'excludedGit': excluded, 'uncollectedPackages': packages,
'conditionValues': conditions,
'scope': 'Union of Linux/Windows x64 minimal Git snapshots; not a linked-target list. '
'No hooks, CIPD/GCS packages, compiler/sysroot downloads or build executed.'}
def export(repo, item, archive, inventory):
listing = run(repo, ['ls-tree', '-rlz', item['revision']])
listing.check_returncode()
entries, total = [], 0
for raw in listing.stdout.split(b'\0'):
if not raw:
continue
metadata, name = raw.split(b'\t', 1)
mode, kind, oid, size = metadata.decode('ascii').split()
name = name.decode('utf-8')
path = PurePosixPath(name)
if path.is_absolute() or '..' in path.parts or '\\' in name:
raise ValueError('Unsafe Git path')
size = int(size) if size != '-' else 0
if size > 512 * 1024**2 or mode not in {'100644', '100755', '120000', '160000'}:
raise ValueError('Unexpected source entry')
total += size
entries.append({'path': name, 'mode': mode, 'type': kind, 'gitObject': oid, 'bytes': size})
if len(entries) > 300000 or total > 4 * 1024**3:
raise ValueError('Source snapshot exceeds collection limits')
prefix = 'pdfium/' + (item['path'] + '/' if item['path'] != '.' else '')
child = subprocess.Popen(GIT + ['--git-dir=' + str(repo), 'cat-file', '--batch'],
env=ENV, stdin=subprocess.PIPE, stdout=subprocess.PIPE,
stderr=subprocess.PIPE)
try:
with archive.open('xb') as raw, gzip.GzipFile(filename='', mode='wb', fileobj=raw, mtime=0) as zipped:
with tarfile.open(fileobj=zipped, mode='w|', format=tarfile.PAX_FORMAT) as tar:
for entry in entries:
if entry['type'] == 'commit':
entry['archived'] = False
continue
child.stdin.write((entry['gitObject'] + '\n').encode()); child.stdin.flush()
header = child.stdout.readline().decode().split()
if header != [entry['gitObject'], 'blob', str(entry['bytes'])]:
raise ValueError('Git blob header differs')
blob = child.stdout.read(entry['bytes'])
if len(blob) != entry['bytes'] or child.stdout.read(1) != b'\n':
raise ValueError('Short Git blob read')
object_hash = hashlib.sha1(b'blob ' + str(len(blob)).encode() + b'\0' + blob).hexdigest()
if object_hash != entry['gitObject']:
raise ValueError('Git object content differs')
entry['sha256'] = hashlib.sha256(blob).hexdigest()
entry['archived'] = True
info = tarfile.TarInfo(prefix + entry['path'])
info.mtime = 0
info.mode = int(entry['mode'][-3:], 8)
if entry['mode'] == '120000':
info.type = tarfile.SYMTYPE
info.linkname = blob.decode('utf-8')
entry['linkTarget'] = info.linkname
tar.addfile(info)
else:
info.size = len(blob)
tar.addfile(info, io.BytesIO(blob))
child.stdin.close()
if child.wait(timeout=60) != 0:
raise ValueError('Git cat-file failed')
finally:
if child.poll() is None:
child.kill(); child.wait()
with inventory.open('x') as stream:
for entry in entries:
stream.write(json.dumps(entry, ensure_ascii=False, sort_keys=True) + '\n')
# Read the entire resulting archive and independently compare payloads.
by_name = {prefix + e['path']: e for e in entries if e['archived']}
with tarfile.open(archive, 'r:gz') as tar:
for member in tar:
entry = by_name.pop(member.name)
if member.issym():
if member.linkname != entry['linkTarget']:
raise ValueError('Archived symlink differs')
else:
with tar.extractfile(member) as stream:
digest = hashlib.file_digest(stream, 'sha256').hexdigest()
if digest != entry['sha256'] or member.size != entry['bytes']:
raise ValueError('Archived blob differs')
if by_name:
raise ValueError('Archive omitted entries')
return {'entries': len(entries), 'sourceBytes': total,
'regularFiles': sum(e['type'] == 'blob' and e['mode'] != '120000' for e in entries),
'symlinks': sum(e['mode'] == '120000' for e in entries),
'gitlinks': [e for e in entries if e['type'] == 'commit'],
'archiveVerifiedAgainstGitBlobs': True}
def collect(item, cache, output):
key = 'root' if item['path'] == '.' else item['path'].replace('/', '--')
directory = output / key
directory.mkdir()
repo = cache / (key + '.git')
# Reuse the previously fetched, verified root; no source checkout is made.
probe = ROOT / '.deps/source-archives/pdfium-probe/repository.git'
if key == 'root' and probe.is_dir():
repo = probe
report = dict(item, repository=str(repo.relative_to(ROOT)), success=False)
try:
if not repo.exists():
subprocess.run(GIT + ['init', '--bare', '--template=', str(repo)], env=ENV,
capture_output=True, check=True, timeout=20)
exists = run(repo, ['cat-file', '-e', item['revision'] + '^{commit}'])
if exists.returncode:
fetched = run(repo, ['fetch', '--depth=1', '--no-tags', item['url'], item['revision']], timeout=900)
(directory / 'fetch.log').write_bytes(fetched.stdout + fetched.stderr)
fetched.check_returncode()
report['fetched'] = True
else:
report['fetched'] = False
commit = run(repo, ['rev-parse', item['revision'] + '^{commit}'])
commit.check_returncode()
if commit.stdout.decode().strip() != item['revision']:
raise ValueError('Commit mismatch')
tree = run(repo, ['rev-parse', item['revision'] + '^{tree}'])
tree.check_returncode()
report['tree'] = tree.stdout.decode().strip()
if item.get('expectedTree') and report['tree'] != item['expectedTree']:
raise ValueError('Root tree differs from saved Gitiles metadata')
checked = run(repo, ['fsck', '--full', '--strict', '--no-reflogs', item['revision']], timeout=300)
(directory / 'fsck.log').write_bytes(checked.stdout + checked.stderr)
checked.check_returncode()
report['gitFsckPassed'] = True
archive = cache / (key + '-' + item['revision'] + '.tar.gz')
inventory = directory / 'files.jsonl'
report.update(export(repo, item, archive, inventory))
report.update(archive=str(archive.relative_to(ROOT)), archiveBytes=archive.stat().st_size,
archiveSha256=sha(archive), inventory=str(inventory.relative_to(ROOT)),
inventorySha256=sha(inventory), success=True)
except Exception as error:
report['error'] = type(error).__name__ + ': ' + str(error)
save(directory / 'report.json', report)
print(json.dumps({'path': item['path'], 'success': report['success'],
'error': report.get('error'), 'bytes': report.get('archiveBytes')}), flush=True)
return report
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--output', type=Path, required=True)
parser.add_argument('--cache', type=Path, default=ROOT / '.deps/source-archives/pdfium')
args = parser.parse_args()
output, cache = args.output.resolve(), args.cache.resolve()
if not output.is_relative_to(ROOT) or not cache.is_relative_to(ROOT):
parser.error('Output/cache must be inside the workspace')
selected = plan()
output.mkdir(parents=True, exist_ok=False); cache.mkdir(parents=True, exist_ok=True)
save(output / 'plan.json', selected)
report = {'startedAt': datetime.now(timezone.utc).isoformat(), 'pinSha256': sha(PINS),
'collectorSha256': sha(Path(__file__)), 'planSha256': sha(output / 'plan.json'),
'gitVersion': subprocess.check_output(['git', '--version'], env=ENV, text=True).strip()}
values = []
with ThreadPoolExecutor(max_workers=3) as pool:
futures = [pool.submit(collect, item, cache, output) for item in selected['selectedGit']]
for future in as_completed(futures):
values.append(future.result())
report.update(success=all(x['success'] for x in values), repositories=sorted(values, key=lambda x: x['path']),
finishedAt=datetime.now(timezone.utc).isoformat(),
completeCorrespondingSources=False, buildReproduced=False,
scope=selected['scope'])
save(output / 'report.json', report)
print(json.dumps({'success': report['success'], 'repositories': len(values)}))
return 0 if report['success'] else 1
if __name__ == '__main__':
raise SystemExit(main())
@@ -0,0 +1,50 @@
#!/usr/bin/env python3
"""Preserve the single nested Git source referenced by the collected FreeType tree."""
import argparse
import configparser
import json
from pathlib import Path
from collect_pdfium import ROOT, collect, run, save, sha
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--output', type=Path, required=True)
parser.add_argument('--cache', type=Path, default=ROOT / '.deps/source-archives/pdfium-gitlink')
args = parser.parse_args()
output, cache = args.output.resolve(), args.cache.resolve()
if not output.is_relative_to(ROOT) or not cache.is_relative_to(ROOT):
parser.error('Output/cache must be inside the workspace')
collection = ROOT / 'tests/results/source-archives/pdfium-git/report.json'
report = json.loads(collection.read_text())
if not report['success']:
raise ValueError('Verified parent collection required')
parent = next(item for item in report['repositories'] if item['path'] == 'third_party/freetype/src')
links = [(item['path'], link) for item in report['repositories'] for link in item['gitlinks']]
if (len(links) != 1 or links[0][0] != parent['path'] or links[0][1]['path'] != 'subprojects/dlg'
or links[0][1]['gitObject'] != '395ccad2c1e0daae535c4d20bb0a3f2424648e17'):
raise ValueError('Unreviewed nested Git source')
config = run(ROOT / parent['repository'], ['show', parent['revision'] + ':.gitmodules'])
config.check_returncode()
modules = configparser.ConfigParser(interpolation=None)
modules.read_string(config.stdout.decode())
source = modules['submodule "dlg"']
if source['path'] != 'subprojects/dlg' or source['url'] != 'https://github.com/nyorain/dlg.git':
raise ValueError('Unexpected pinned .gitmodules mapping')
output.mkdir(parents=True, exist_ok=False); cache.mkdir(parents=True, exist_ok=True)
(output / 'freetype.gitmodules').write_bytes(config.stdout)
item = {'path': parent['path'] + '/' + source['path'], 'url': source['url'],
'revision': links[0][1]['gitObject'], 'condition': 'Referenced by collected FreeType gitlink'}
result = collect(item, cache, output)
success = result['success'] and not result.get('gitlinks')
save(output / 'report.json', {'success': success, 'collectorSha256': sha(Path(__file__)),
'parentCollectionSha256': sha(collection), 'parentRevision': parent['revision'],
'gitmodulesSha256': sha(output / 'freetype.gitmodules'), 'repositories': [result],
'scope': 'Nested source preserved, not evidence that dlg is linked into the distributed PDFium. '
'No source/build scripts or hooks executed.'})
return 0 if success else 1
if __name__ == '__main__':
raise SystemExit(main())
@@ -0,0 +1,260 @@
#!/usr/bin/env python3
"""Extract the tested Ubuntu Qt SDK's embedded Chromium SBOM notice texts."""
import argparse
import ast
import hashlib
import json
from pathlib import Path, PurePosixPath
import subprocess
import tarfile
import threading
ROOT = Path(__file__).resolve().parents[2]
RUNTIME = ROOT / 'tests/results/ui-final/ubuntu/installed-runtime-final'
QT_SOURCE = ROOT / '.deps/source-archives/qt-6.11.2-regular-files'
ARCHIVE_NAME = '6.11.2-0-202608131118qtwebengine-Linux-RHEL_9_6-GCC-Linux-RHEL_9_6-X86_64.7z'
ARCHIVE_SHA256 = '9cbfd85900e85b95fa11926b41818addfc2a96361f62af567be430607ba6b67e'
METADATA_NAMES = ['qtwebengine-6.11.2.spdx.json', 'qtwebengine-chromium-webengine-6.11.2.spdx.json',
'qtwebengine-chromium-webengine-6.11.2.spdx']
def sha(path):
with path.open('rb') as stream:
return hashlib.file_digest(stream, 'sha256').hexdigest()
def write(path, data):
path.write_text(json.dumps(data, indent=2, ensure_ascii=False) + '\n')
def relative(name):
path = PurePosixPath(name)
if path.is_absolute() or '..' in path.parts or '\\' in name:
raise ValueError('Unsafe input path')
return str(path)
def inspect_sdk_archive(archive, output):
"""Stream archive conversion, never extracting links or executing payloads."""
rows, names, total = [], set(), 0
with (output / 'archive-reader.log').open('wb') as error_log:
process = subprocess.Popen(['bsdtar', '-cf', '-', '--format=pax', '@' + str(archive)],
stdout=subprocess.PIPE, stderr=error_log)
timer = threading.Timer(240, process.kill)
timer.start()
try:
with tarfile.open(fileobj=process.stdout, mode='r|') as tar:
for member in tar:
name = relative(member.name)
if name in names or len(names) >= 10000:
raise ValueError('Duplicate/excessive SDK entries')
names.add(name)
if member.isdir():
continue
item = {'path': name, 'mode': member.mode}
if member.issym() or member.islnk():
item.update(type='symlink' if member.issym() else 'hardlink', target=member.linkname)
elif member.isfile():
total += member.size
if member.size > 512 * 1024**2 or total > 4 * 1024**3:
raise ValueError('SDK exceeds inspection size limits')
hashes = {key: hashlib.new(key) for key in ('sha1', 'sha256')}
size = 0
with tar.extractfile(member) as stream:
for chunk in iter(lambda: stream.read(1024 * 1024), b''):
size += len(chunk)
for digest in hashes.values():
digest.update(chunk)
if size != member.size:
raise ValueError('Short archive member')
item.update(type='file', bytes=size, **{key: value.hexdigest() for key, value in hashes.items()})
else:
raise ValueError('Unexpected SDK archive entry type')
rows.append(item)
if process.wait(timeout=30):
raise ValueError('SDK archive inspection failed')
finally:
timer.cancel()
if process.poll() is None:
process.kill(); process.wait()
rows.sort(key=lambda x: x['path'])
write(output / 'sdk-archive-inventory.json', rows)
return {row['path']: row for row in rows}, total
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--output', type=Path, required=True)
args = parser.parse_args()
output = args.output.resolve()
if not output.is_relative_to(ROOT):
parser.error('Output must be inside the workspace')
output.mkdir(parents=True, exist_ok=False)
(output / 'notices').mkdir()
runtime_path = RUNTIME / 'runtime.json'
runtime = json.loads(runtime_path.read_text())
metadata = {}
for name in METADATA_NAMES:
entry = next(e for e in runtime['notices'] if e['path'] == 'qt-sdk:sbom/' + name)
path = RUNTIME / entry['copiedPath']
if path.stat().st_size != entry['size'] or sha(path) != entry['sha256']:
raise ValueError('Saved SDK metadata changed')
metadata[name] = dict(entry, localPath=str(path.relative_to(ROOT)))
main_sbom = json.loads((ROOT / metadata[METADATA_NAMES[0]]['localPath']).read_text())
chromium = json.loads((ROOT / metadata[METADATA_NAMES[1]]['localPath']).read_text())
external = next(e for e in main_sbom['externalDocumentRefs'] if 'qtwebengine-chromium-webengine' in e['externalDocumentId'])
external_bytes = (ROOT / metadata[METADATA_NAMES[2]]['localPath']).read_bytes()
if (external['checksum']['algorithm'] != 'SHA1'
or hashlib.sha1(external_bytes).hexdigest() != external['checksum']['checksumValue']
or external['spdxDocument'] != chromium['documentNamespace']):
raise ValueError('Qt module SBOM does not bind the Chromium SBOM')
archive = ROOT / 'build-ubuntu-vm/sdk-downloads' / ARCHIVE_NAME
if sha(archive) != ARCHIVE_SHA256:
raise ValueError('Qt SDK archive differs from fixed downloaded input')
inventory, total_bytes = inspect_sdk_archive(archive, output)
for name, entry in metadata.items():
if inventory['sbom/' + name]['sha256'] != entry['sha256']:
raise ValueError('SDK metadata differs from original archive')
file_checks = []
for file in main_sbom['files']:
name = relative(file['fileName'])
entry = inventory.get(name)
if not entry or entry['type'] != 'file':
raise ValueError('SBOM file absent from original archive: ' + name)
for checksum in file['checksums']:
algorithm = checksum['algorithm'].lower()
if algorithm not in {'sha1', 'sha256'} or entry[algorithm] != checksum['checksumValue']:
raise ValueError('SBOM file checksum differs: ' + name)
file_checks.append({'path': name, 'spdxId': file['SPDXID'], 'sha1': entry['sha1'], 'sha256': entry['sha256']})
runtime_checks = []
for entry in runtime['files']:
resolved = entry['resolvedPath']
if not resolved.startswith('qt-sdk:'):
continue
name = relative(resolved.removeprefix('qt-sdk:'))
if name not in inventory:
continue
original = inventory[name]
if original.get('sha256') != entry['sha256'] or original.get('bytes') != entry['size']:
raise ValueError('SDK archive differs from tested Ubuntu runtime: ' + name)
runtime_checks.append({'path': entry['path'], 'resolvedPath': resolved, 'sha256': entry['sha256']})
if not any(e['resolvedPath'] == 'qt-sdk:lib/libQt6WebEngineCore.so.6.11.2' for e in runtime_checks):
raise ValueError('QtWebEngineCore runtime binding is required')
packages = {p['SPDXID']: p for p in chromium['packages']}
licenses = {p['licenseId']: p for p in chromium['hasExtractedLicensingInfos']}
if len(packages) != len(chromium['packages']) or len(licenses) != len(chromium['hasExtractedLicensingInfos']):
raise ValueError('Duplicate SPDX IDs')
# The SDK's document checksum/namespace can match even when its package
# relationship names a nonexistent ID. Preserve that upstream inconsistency.
external_package_refs = []
for relationship in main_sbom['relationships']:
prefix = external['externalDocumentId'] + ':'
target = relationship['relatedSpdxElement']
if target.startswith(prefix):
target_id = target[len(prefix):]
external_package_refs.append(dict(relationship, targetIdExists=target_id in packages))
for relationship in chromium['relationships']:
if (relationship['spdxElementId'] not in packages
or relationship['relatedSpdxElement'] not in packages
or relationship['relationshipType'] != 'CONTAINS'):
raise ValueError('Unreviewed SPDX relationship')
without_text = []
for package in packages.values():
license_id = package['licenseConcluded']
if license_id not in licenses:
if license_id not in {'BSD-3-Clause', 'MIT'}:
raise ValueError('Unresolved SPDX license reference')
without_text.append(package)
source_report = json.loads((ROOT / 'tests/results/source-archives/qt-inspection/report.json').read_text())
source_inventory = ROOT / 'tests/results/source-archives/qt-inspection/files.jsonl'
if sha(source_inventory) != source_report['inventorySha256']:
raise ValueError('Qt source inventory changed')
helpers = ['qtwebengine/src/3rdparty/chromium/tools/licenses/sbom.py',
'qtwebengine/cmake/QtWebEngineSbomHelpers.cmake']
source_names = set(helpers)
for license in licenses.values():
refs = license['crossRefs']
if len(refs) != 1 or not refs[0]['url'].startswith(('/chromium/', '/gn/')):
raise ValueError('Unexpected source license reference')
source_names.add('qtwebengine/src/3rdparty/' + relative(refs[0]['url'][1:]))
sources = {}
for line in source_inventory.open():
entry = json.loads(line)
if entry['path'] in source_names:
if sha(QT_SOURCE / entry['path']) != entry['sha256']:
raise ValueError('Qt source notice/helper changed')
sources[entry['path']] = entry
if sources.keys() != source_names:
raise ValueError('Notice source missing from verified source archive')
rows, combined = [], ['Qt WebEngine 6.11.2 — Chromium notices contained in the Linux Qt SDK SBOM\n']
for license_id, license in sorted(licenses.items()):
text = license['extractedText'].encode('utf-8')
source = 'qtwebengine/src/3rdparty/' + license['crossRefs'][0]['url'][1:]
if not text or text != (QT_SOURCE / source).read_bytes():
raise ValueError('SBOM license text differs from verified source archive')
digest = hashlib.sha256(text).hexdigest()
target = output / 'notices' / (digest + '.txt')
if not target.exists():
target.write_bytes(text)
consumers = [p['SPDXID'] for p in packages.values() if p['licenseConcluded'] == license_id]
rows.append({'licenseId': license_id, 'source': source, 'sha256': digest, 'bytes': len(text),
'file': str(target.relative_to(output)), 'packageIds': consumers})
combined.extend(['\n' + '=' * 72 + '\n' + license['name'] + '\n',
'Source: ' + license['crossRefs'][0]['url'] + '\n\n', license['extractedText']])
bundle = output / 'THIRD_PARTY_NOTICES.sdk-extract.txt'
bundle.write_text(''.join(combined), encoding='utf-8')
# Preserve the generator's explicit limitations as data without importing it.
syntax = ast.parse((QT_SOURCE / helpers[0]).read_text())
skip_names = {'DIRECTORIES_TO_SKIP_BECAUSE_THEY_HAVE_VARIOUS_PARSING_ISSUES',
'PACKAGES_TO_OVERRIDE_LICENSE_FILE_WITH_ID'}
generator_limits = {}
for node in syntax.body:
if not isinstance(node, ast.Assign) or not isinstance(node.targets[0], ast.Name):
continue
name = node.targets[0].id
if name not in skip_names:
continue
values = []
for value in node.value.elts:
if isinstance(value, ast.Constant):
values.append(value.value)
elif isinstance(value, ast.Call) and ast.unparse(value.func) == 'os.path.join':
values.append('/'.join(ast.literal_eval(v) for v in value.args))
else:
raise ValueError('Unexpected generator limitation declaration')
generator_limits[name] = values
arch_sbom = Path('/usr/lib/qt6/sbom/qtwebengine-6.11.2.spdx')
arch_incomplete = 'not listing all of its consumed 3rd party dependencies' in arch_sbom.read_text()
if not arch_incomplete:
raise ValueError('Arch SBOM scope changed; reassess separately')
(output / 'arch-qtwebengine.spdx').write_bytes(arch_sbom.read_bytes())
report = {'success': True, 'collectorSha256': sha(Path(__file__)), 'runtimeRecordSha256': sha(runtime_path),
'sdkArchive': {'path': str(archive.relative_to(ROOT)), 'sha256': ARCHIVE_SHA256,
'bytes': archive.stat().st_size, 'regularPayloadBytes': total_bytes,
'entries': len(inventory)},
'sdkArchiveInventorySha256': sha(output / 'sdk-archive-inventory.json'),
'metadata': metadata, 'externalSbomReference': external, 'sbomFileComparisons': file_checks,
'externalPackageReferences': external_package_refs,
'externalPackageReferencesValid': bool(external_package_refs) and all(
e['targetIdExists'] for e in external_package_refs),
'testedRuntimeComparisons': runtime_checks, 'packages': chromium['packages'],
'relationships': chromium['relationships'], 'notices': rows,
'noticeBundleSha256': sha(bundle), 'uniqueNoticeFiles': len(list((output / 'notices').glob('*.txt'))),
'packagesWithoutEmbeddedNoticeText': without_text, 'generatorLimits': generator_limits,
'generatorSources': {name: sources[name]['sha256'] for name in helpers},
'qtSourceArchiveSha256': source_report['archiveSha256'],
'archSbom': {'path': str(arch_sbom), 'sha256': sha(arch_sbom), 'explicitIncompleteNotice': True},
'completeChromiumNotices': False, 'completeCorrespondingSources': False,
'scope': 'All embedded notice texts in the tested Ubuntu Qt SDK Chromium WebEngine SBOM, '
'bound to its original archive, runtime fingerprint and Qt source archive. '
'Not the Arch configuration or a complete license/linked-component verdict. '
'Unresolved external package IDs, generator skips, identifier-only entries '
'and omitted patent files remain distinct.'}
write(output / 'report.json', report)
print(json.dumps({'success': True, 'noticeEntries': len(rows), 'uniqueNoticeFiles': report['uniqueNoticeFiles'],
'sbomFilesMatched': len(file_checks), 'testedRuntimeFilesMatched': len(runtime_checks),
'packagesWithoutEmbeddedText': len(without_text)}))
if __name__ == '__main__':
main()
@@ -0,0 +1,117 @@
#!/usr/bin/env python3
"""Recover two explicitly declared Patent files omitted by the fixed Qt SDK SBOM."""
import argparse
import hashlib
import json
from pathlib import Path
import re
ROOT = Path(__file__).resolve().parents[2]
BASE = ROOT / 'tests/results/source-archives/qt-sdk-notices-final/report.json'
BASE_SHA = 'cea0a2f80ad50b5519fbad94c1dd2eed55f5ffa34e83d1360ce8da6902a845a0'
INVENTORY_SHA = '43dc4aa3cf0d970fc97a9db2555f0025429e9bd31823cb794fd5526496d1a9e0'
SOURCE = ROOT / '.deps/source-archives/qt-6.11.2-regular-files'
PREFIX = 'qtwebengine/src/3rdparty/chromium/'
PATENTS = {
'libwebm': ('source/LICENSE.TXT', 'source/PATENTS.TXT'),
'libwebp': ('src/COPYING', 'src/PATENTS'),
}
ID_ONLY = {
'mini-chromium': ('third_party/crashpad/crashpad/third_party/mini_chromium/README.crashpad', 'Shipped in Chromium'),
'libdrm': ('third_party/libdrm/README.chromium', 'Shipped'),
}
def sha(path):
with path.open('rb') as stream:
return hashlib.file_digest(stream, 'sha256').hexdigest()
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--output', type=Path, required=True)
args = parser.parse_args()
output = args.output.resolve()
if not output.is_relative_to(ROOT):
parser.error('Output must be inside the workspace')
assert sha(BASE) == BASE_SHA
base = json.loads(BASE.read_text())
inventory = ROOT / 'tests/results/source-archives/qt-inspection/files.jsonl'
assert sha(inventory) == INVENTORY_SHA
names = {PREFIX + 'tools/licenses/sbom.py'}
for name, files in PATENTS.items():
names.update(PREFIX + 'third_party/' + name + '/' + tail for tail in (*files, 'README.chromium'))
names.update(PREFIX + row[0] for row in ID_ONLY.values())
sources = {}
for line in inventory.open():
row = json.loads(line)
if row['path'] in names:
assert row['path'] not in sources
source = SOURCE / row['path']
assert source.is_file() and not source.is_symlink()
assert source.stat().st_size == row['bytes'] and sha(source) == row['sha256']
sources[row['path']] = row
assert sources.keys() == names
assert sources[PREFIX + 'tools/licenses/sbom.py']['sha256'] == base['generatorSources'][PREFIX + 'tools/licenses/sbom.py']
packages = {p['SPDXID']: p for p in base['packages']}
output.mkdir(parents=True, exist_ok=False)
copied, notices, id_only = [], [], []
def copy(name, kind):
source = sources[name]
target = 'sources/' + name.removeprefix(PREFIX)
path = output / target
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes((SOURCE / name).read_bytes())
assert sha(path) == source['sha256']
copied.append(dict(file=target, source=name, sha256=source['sha256'], bytes=source['bytes'], kind=kind))
return target
def fields(name):
return dict(re.findall(r'^([^:\n]+): (.*)$', (SOURCE / name).read_text(), re.M))
for name, (license_file, patent_file) in PATENTS.items():
directory = PREFIX + 'third_party/' + name + '/'
metadata = directory + 'README.chromium'
data = fields(metadata)
assert data['License'] == 'BSD-3-Clause, Patent' and data['Shipped'] == 'yes'
assert data['License File'] == license_file + ', ' + patent_file
package_id = 'SPDXRef-Package-QtWebEngine-Chromium-WebEngine-' + name
package = packages[package_id]
assert package['comment'].splitlines()[0] == 'Location within source: third_party/' + name
embedded = next(n for n in base['notices'] if package_id in n['packageIds'])
assert embedded['source'] == directory + license_file
assert embedded['sha256'] == sources[directory + license_file]['sha256']
assert not any(n['source'] == directory + patent_file for n in base['notices'])
metadata_file = copy(metadata, 'upstream-package-metadata')
notice_file = copy(directory + patent_file, 'declared-patent-notice')
notices.append(dict(packageId=package_id, revision=data['Revision'], licenseDeclared=data['License'],
metadataFile=metadata_file, noticeFile=notice_file,
existingLicenseSha256=embedded['sha256'], reason='Second declared file omitted by SDK generator Patent branch'))
for name, (relative, shipped_field) in ID_ONLY.items():
metadata = PREFIX + relative
data = fields(metadata)
assert data[shipped_field] == 'no'
package_id = 'SPDXRef-Package-QtWebEngine-Chromium-WebEngine-' + name
package = next(p for p in base['packagesWithoutEmbeddedNoticeText'] if p['SPDXID'] == package_id)
assert data['License'] == package['licenseConcluded']
id_only.append(dict(packageId=package_id, metadataFile=copy(metadata, 'upstream-package-metadata'),
declaredShippedField=shipped_field, declaredShippedValue='no',
licenseId=package['licenseConcluded'], noticeAdded=False,
linkedInSdk='unverified; upstream metadata is not a build-graph proof'))
report = dict(schemaVersion=1, success=True, collectorSha256=sha(Path(__file__)),
baseNoticeReportSha256=BASE_SHA, sourceInventorySha256=INVENTORY_SHA,
qtSourceArchiveSha256=base['qtSourceArchiveSha256'], sdkArchiveSha256=base['sdkArchive']['sha256'],
runtimeComparisons=base['testedRuntimeComparisons'], files=sorted(copied, key=lambda x:x['file']),
sourceChecks=sorted(sources.values(), key=lambda x:x['path']), notices=notices, idOnlyPackages=id_only,
sourceSdkExternalPackageReferencesValid=base['externalPackageReferencesValid'],
completeChromiumNotices=False, completeCorrespondingSources=False, publicReleaseApproved=False,
scope='Two upstream-declared Patent files recovered from the fixed Qt 6.11.2 source archive. '
'Four upstream metadata files retained. Original SBOM is unchanged. '
'Does not resolve the SDK build graph, generator skips or external package-ID inconsistency.')
(output / 'report.json').write_text(json.dumps(report, indent=2, sort_keys=True) + '\n')
print(json.dumps({'success': True, 'additionalNotices': len(notices), 'metadataFiles': len(copied)-len(notices)}))
if __name__ == '__main__':
main()
+98
View File
@@ -0,0 +1,98 @@
#!/usr/bin/env python3
"""Collect pinned public source archives; never execute downloaded code."""
import argparse
from concurrent.futures import ThreadPoolExecutor
from datetime import datetime, timezone
import hashlib
import json
from pathlib import Path
import time
import urllib.parse
import urllib.request
ROOT = Path(__file__).resolve().parents[2]
PINS = Path(__file__).with_name('pins.json')
HOSTS = {'download.qt.io', 'ftp.jaist.ac.jp', 'github.com', 'codeload.github.com'}
class Redirects(urllib.request.HTTPRedirectHandler):
def redirect_request(self, req, fp, code, msg, headers, newurl):
parsed = urllib.parse.urlparse(newurl)
if parsed.scheme != 'https' or parsed.hostname not in HOSTS:
raise ValueError('Source archive redirect left the pinned HTTPS providers')
return super().redirect_request(req, fp, code, msg, headers, newurl)
def fetch(name, pin, output, cache):
result = {'name': name, 'pin': pin, 'success': False,
'retrievedAtUtc': datetime.now(timezone.utc).isoformat()}
target = cache / pin['name']; partial = target.with_name(target.name + '.part')
before = time.monotonic()
try:
if target.exists() or partial.exists():
raise ValueError('Refusing to replace an existing archive or partial download')
opener = urllib.request.build_opener(Redirects())
if 'metadataUrl' in pin:
with opener.open(pin['metadataUrl'], timeout=30) as source:
metadata = source.read(128 * 1024 + 1)
if len(metadata) > 128 * 1024:
raise ValueError('Metadata exceeded its finite limit')
if not all(value.encode() in metadata for value in pin['hashes'].values()):
raise ValueError('Published Qt checksum differs from the pin')
result['metadataSha256'] = hashlib.sha256(metadata).hexdigest()
# Mirror pages may contain the requesting IP. Keep the digest and
# matched public fields rather than copying that incidental data.
result['publishedHashesMatched'] = True
else:
metadata = (ROOT / pin['metadataFile']).read_bytes()
if not all(value.encode() in metadata for value in pin['hashes'].values()):
raise ValueError('Stored recipe differs from the pinned checksums')
result['metadataSha256'] = hashlib.sha256(metadata).hexdigest()
limit = pin.get('bytes', pin.get('maximumBytes'))
hashes = {algorithm: hashlib.new(algorithm) for algorithm in {*pin['hashes'], 'sha256'}}
count, last = 0, before
with opener.open(pin['url'], timeout=30) as source, partial.open('xb') as file:
result['resolvedUrl'] = source.url
while block := source.read(1024 * 1024):
count += len(block)
if count > limit or time.monotonic() - before > 900:
raise ValueError('Download exceeded size or time limit')
file.write(block)
for digest in hashes.values():
digest.update(block)
if time.monotonic() - last > 10:
print(f'{name}: {count / 1048576:.1f} MiB received', flush=True)
last = time.monotonic()
result.update(bytes=count, hashes={algorithm: digest.hexdigest() for algorithm, digest in hashes.items()})
if pin.get('bytes', count) != count or any(result['hashes'][algorithm] != digest for algorithm, digest in pin['hashes'].items()):
raise ValueError('Archive size or checksum mismatch; partial retained as unverified')
partial.rename(target)
result.update(success=True, archive=str(target.relative_to(ROOT)))
except Exception as error:
result['error'] = str(error)
result['elapsedSeconds'] = time.monotonic() - before
(output / (name + '.json')).write_text(json.dumps(result, indent=2) + '\n')
print(json.dumps({'name': name, 'success': result['success'], 'error': result.get('error')}), flush=True)
return result['success']
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--output', type=Path, required=True)
parser.add_argument('--cache', type=Path, required=True)
parser.add_argument('--artifact', action='append', choices=['qt-everywhere', 'tomlplusplus'])
args = parser.parse_args()
output, cache = args.output.resolve(), args.cache.resolve()
if not output.is_relative_to(ROOT) or not cache.is_relative_to(ROOT):
parser.error('Keep this task’s evidence and archives inside the workspace')
output.mkdir(parents=True, exist_ok=False); cache.mkdir(parents=True, exist_ok=True)
pins = json.loads(PINS.read_text())
names = args.artifact or list(pins)
with ThreadPoolExecutor(max_workers=2) as executor:
futures = [executor.submit(fetch, name, pins[name], output, cache) for name in names]
successes = [future.result() for future in futures]
return 0 if all(successes) else 1
if __name__ == '__main__':
raise SystemExit(main())
+130
View File
@@ -0,0 +1,130 @@
#!/usr/bin/env python3
"""Inspect the verified Qt release archive and retain regular source files safely."""
import argparse
import hashlib
import json
from pathlib import Path, PurePosixPath
import re
import tarfile
import time
import urllib.parse
ROOT = Path(__file__).resolve().parents[2]
def sha(path):
with path.open('rb') as f:
return hashlib.file_digest(f, 'sha256').hexdigest()
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--archive', type=Path, required=True)
parser.add_argument('--tree', type=Path, required=True)
parser.add_argument('--output', type=Path, required=True)
args = parser.parse_args()
archive, tree, out = args.archive.resolve(), args.tree.resolve(), args.output.resolve()
if any(not path.is_relative_to(ROOT) for path in [archive, tree, out]):
parser.error('All paths must remain in this workspace')
pin = json.loads(Path(__file__).with_name('pins.json').read_text())['qt-everywhere']
if archive.stat().st_size != pin['bytes'] or sha(archive) != pin['hashes']['sha256']:
parser.error('Archive does not match the verified release pin')
tree.mkdir(parents=True, exist_ok=False); out.mkdir(parents=True, exist_ok=False)
comparison_root = ROOT / 'tests/results/source-correspondence/arch/upstream-metadata'
expected = {}
for row in json.loads((comparison_root / 'version-sources.json').read_text()):
url = urllib.parse.urlparse(row['url']).path
if '/qtwebengine.git/plain/' in url:
relative = 'qtwebengine/' + url.split('/plain/', 1)[1]
elif '/qtwebengine-chromium.git/plain/' in url:
relative = 'qtwebengine/src/3rdparty/' + url.split('/plain/', 1)[1]
else:
continue
if sha(comparison_root / row['file']) != row['sha256']:
raise ValueError('Stored fixed-revision comparison input changed')
expected[relative] = row
binaries = {name: sha(ROOT / 'build' / name) for name in ['docview', 'docview-pdf-worker', 'docview-archive-worker']}
report = {'success': False, 'archive': str(archive.relative_to(ROOT)), 'archiveSha256': pin['hashes']['sha256'],
'archiveBytes': pin['bytes'], 'sourceTree': str(tree.relative_to(ROOT)),
'scope': 'Official Qt 6.11.2 release sources and selected metadata/notices; not complete linked-component attribution, a rebuild, license acceptance or legal compliance certification'}
files = size = members = notice_bytes = 0
notices, links, comparisons, module_counts, versions = [], [], {}, {}, {}
seen = set(); last = time.monotonic()
try:
with tarfile.open(archive, 'r|xz') as stream, (out / 'files.jsonl').open('w') as inventory:
for member in stream:
members += 1
if members > 600000 or member.size > 1024**3 or member.size < 0:
raise ValueError('Archive exceeded finite member limits')
path = PurePosixPath(member.name)
if path.is_absolute() or '..' in path.parts or not path.parts or path.parts[0] != 'qt-everywhere-src-6.11.2' or '\\' in member.name:
raise ValueError('Unexpected source member path')
relative = PurePosixPath(*path.parts[1:])
if member.isdir():
continue
name = str(relative)
if name in seen or name == '.':
raise ValueError('Duplicate or empty source file path')
seen.add(name)
if member.issym() or member.islnk():
links.append({'path': name, 'target': member.linkname, 'kind': 'symlink' if member.issym() else 'hardlink'})
continue # Keep links in the original archive; never follow them.
if not member.isfile():
raise ValueError('Non-file source entry')
size += member.size
if size > 20 * 1024**3:
raise ValueError('Source tree exceeded 20 GiB limit')
destination = tree / name
destination.parent.mkdir(parents=True, exist_ok=True)
digest = hashlib.sha256(); count = 0
with stream.extractfile(member) as source, destination.open('xb') as target:
while block := source.read(1024 * 1024):
target.write(block); digest.update(block); count += len(block)
if count != member.size:
raise ValueError('Source member was truncated')
row = {'path': name, 'bytes': count, 'sha256': digest.hexdigest(), 'archiveMode': member.mode}
inventory.write(json.dumps(row) + '\n'); files += 1
module = relative.parts[0] if len(relative.parts) > 1 else '(root)'
module_counts[module] = module_counts.get(module, 0) + 1
if name in expected:
comparisons[name] = {**row, 'expectedSha256': expected[name]['sha256'],
'matches': row['sha256'] == expected[name]['sha256'], 'sourceUrl': expected[name]['url']}
if relative.name == '.cmake.conf' and len(relative.parts) == 2:
text = destination.read_text(errors='replace')
versions[module] = re.findall(r'QT_REPO_MODULE_VERSION\s+"([^"]+)"', text)
# Keep a named-file subset; README.chromium and attribution
# metadata may refer to further texts. Do not call this complete.
if (re.fullmatch(r'(licen[cs]e|copying|copyright|notice)([._-].*)?', relative.name, re.I)
or relative.name in ('README.chromium', 'qt_attributions.json', 'REUSE.toml')
or 'LICENSES' in relative.parts):
if count > 8 * 1024**2 or notice_bytes + count > 128 * 1024**2:
raise ValueError('Notice metadata exceeded finite limits')
notice_bytes += count; notices.append(row)
if time.monotonic() - last > 15:
print(f'Qt source: {files} files / {size / 1048576:.1f} MiB inspected', flush=True)
last = time.monotonic()
report['fixedSourceComparisons'] = comparisons
missing = sorted(set(expected) - set(comparisons))
report['missingComparisons'] = missing
if missing or not all(row['matches'] for row in comparisons.values()):
raise ValueError('Release source differs from a fixed-revision comparison input')
report['success'] = True
except Exception as error:
report['error'] = str(error)
finally:
(out / 'notices-index.json').write_text(json.dumps(notices, indent=2) + '\n')
report.update(regularFiles=files, regularBytes=size, members=members, archivedLinksNotMaterialized=links,
moduleFileCounts=module_counts, moduleVersionDeclarations=versions,
noticeAndMetadataFiles=len(notices), noticeAndMetadataBytes=notice_bytes,
completeChromiumNotices=False, completeCorrespondingSources=False,
inventorySha256=sha(out / 'files.jsonl'), noticesIndexSha256=sha(out / 'notices-index.json'),
productionBinaries=binaries,
productionBinariesUnchanged=all(sha(ROOT / 'build' / name) == digest for name, digest in binaries.items()))
report['success'] &= report['productionBinariesUnchanged']
(out / 'report.json').write_text(json.dumps(report, indent=2) + '\n')
print(json.dumps({key: report.get(key) for key in ['success', 'regularFiles', 'regularBytes', 'noticeAndMetadataFiles', 'error']}))
return 0 if report['success'] else 1
if __name__ == '__main__':
raise SystemExit(main())
+21
View File
@@ -0,0 +1,21 @@
{
"qt-everywhere": {
"name": "qt-everywhere-src-6.11.2.tar.xz",
"url": "https://ftp.jaist.ac.jp/pub/qtproject/archive/qt/6.11/6.11.2/single/qt-everywhere-src-6.11.2.tar.xz",
"metadataUrl": "https://download.qt.io/archive/qt/6.11/6.11.2/single/qt-everywhere-src-6.11.2.tar.xz.mirrorlist",
"bytes": 1019661552,
"hashes": {
"sha256": "6dcfbca271d76a6502741a2c0dc6fc98ef7dd0b7b4cfd0abcebb285a86a26f33"
}
},
"tomlplusplus": {
"name": "tomlplusplus-3.4.0.tar.gz",
"url": "https://github.com/marzer/tomlplusplus/archive/refs/tags/v3.4.0.tar.gz",
"metadataFile": "tests/results/source-correspondence/arch/packages/tomlplusplus/PKGBUILD",
"maximumBytes": 16777216,
"hashes": {
"sha512": "c227fc8147c9459b29ad24002aaf6ab2c42fac22ea04c1c52b283a0172581ccd4527b33c1931e0ef0d1db6b6a53f9e9882c6d4231c7f3494cf070d0220741aa5",
"blake2b": "9495ccd78707ced11744eab7c1c0bf0c0c28e283d186195bb48d1059bae7eb1a874bc964b0fc45210fd73ffd7485ecf3e1159da227d0e1c8ff249e79c08eecf0"
}
}
}
@@ -0,0 +1,118 @@
#!/usr/bin/env python3
"""Bind a notice-only Ubuntu package update to the existing tested executables."""
from datetime import datetime, timezone
import hashlib
import json
from pathlib import Path
ROOT = Path(__file__).resolve().parents[2]
RESULTS = ROOT / 'tests/results/qt-notice-supplement'
def sha(path):
with path.open('rb') as stream:
return hashlib.file_digest(stream, 'sha256').hexdigest()
def read(name):
return json.loads((RESULTS / name).read_text())
def main():
prior = read('prior-validation-record.json')
assert sha(RESULTS / 'prior-validation-record.json') == '25bc879e35e02b70f7bd1cf2afca293896afb247163fa5b1f81020ec5bd3359b'
for name, digest in prior['evidenceSha256'].items():
assert sha(ROOT / name) == digest, name
assert sha(RESULTS / 'prior-packager.py') == prior['ubuntuPackage']['packagerSha256']
assert sha(RESULTS / 'prior-record-validation.py') == prior['validationToolsSha256']['tests/portal/record_validation.py']
current_sources = {str(path.relative_to(ROOT)): sha(path) for directory in ('src', 'qml', 'cmake', 'resources')
for path in (ROOT / directory).rglob('*') if path.is_file() and '__pycache__' not in path.parts}
current_sources.update({name: sha(ROOT / name) for name in ('CMakeLists.txt', 'resources.qrc')})
assert current_sources == prior['sourceSha256']
for name, digest in prior['builds']['arch']['binaries'].items():
assert sha(ROOT / 'build' / name) == digest
supplement = read('collection/report.json')
repeat = read('collection-repeat/report.json')
assert supplement == repeat and supplement['success']
assert len(supplement['files']) == 6 and len(supplement['notices']) == 2
assert supplement['collectorSha256'] == sha(ROOT / 'tests/source_archives/collect_qt_sdk_supplement.py')
assert supplement['baseNoticeReportSha256'] == sha(ROOT / 'tests/results/source-archives/qt-sdk-notices-final/report.json')
for row in supplement['files']:
assert sha(RESULTS / 'collection' / row['file']) == row['sha256']
assert sha(RESULTS / 'collection-repeat' / row['file']) == row['sha256']
package = read('package/report.json')
repeated = read('package/repeat-report.json')
verified = read('package/verification.json')
manifest = read('package/package-manifest.json')
assert package['success'] and repeated['success'] and verified['success']
assert package['archiveSha256'] == repeated['archiveSha256'] == verified['archiveSha256']
assert package['archiveSha256'] == sha(ROOT / 'build-ubuntu-vm/packages' / package['archive'])
assert package['manifestSha256'] == verified['packageManifestSha256'] == sha(RESULTS / 'package/package-manifest.json')
assert package['packagerSha256'] == sha(ROOT / 'tools/package_ubuntu.py')
assert package['collectorSha256'] == sha(ROOT / 'tools/collect_ubuntu_validation_runtime.py')
assert package['runtimeRecordSha256'] == sha(RESULTS / 'package/runtime/runtime.json')
assert package['executableSha256'] == prior['ubuntuPackage']['executableSha256']
assert package['executableSha256'] == prior['builds']['ubuntu']['installedBinaries']
assert verified['filesVerified'] == package['payloadFiles'] == len(manifest['files']) + 1
assert verified['allSizesModesAndHashesMatch'] and verified['rootOwnership']
assert package['sdkSupplementReportSha256'] == sha(RESULTS / 'collection/report.json')
assert manifest['sdkSupplementReportSha256'] == package['sdkSupplementReportSha256']
old_manifest = json.loads((ROOT / 'tests/results/portal/ubuntu-package/package-manifest.json').read_text())
old_rows = {r['path']: r for r in old_manifest['files']}
new_rows = {r['path']: r for r in manifest['files']}
added = sorted(new_rows.keys() - old_rows.keys())
removed = sorted(old_rows.keys() - new_rows.keys())
changed = sorted(name for name in old_rows.keys() & new_rows.keys() if old_rows[name] != new_rows[name])
assert not removed and len(added) == 7
assert all(name.startswith('opt/docview/share/doc/docview/third-party/qt-sdk-supplement/') for name in added)
assert changed == ['DEBIAN/control', 'opt/docview/share/doc/docview/UBUNTU-PACKAGE.txt']
for row in supplement['files']:
name = 'opt/docview/share/doc/docview/third-party/qt-sdk-supplement/' + row['file']
assert new_rows[name]['sha256'] == row['sha256'] and new_rows[name]['size'] == row['bytes']
lifecycle = {phase: read('package-vm/' + phase + '/report.json') for phase in ('install', 'smoke', 'remove')}
assert all(row['success'] and row['archiveSha256'] == package['archiveSha256'] for row in lifecycle.values())
for phase, report in lifecycle.items():
assert report['runnerSha256'] == sha(ROOT / 'tests/ubuntu_vm/clean_package.py')
for command in report['commands']:
assert command['exitCode'] == 0
assert sha(RESULTS / 'package-vm' / phase / (command['name'] + '.log')) == command['logSha256']
assert lifecycle['install']['executables'] == package['executableSha256']
assert lifecycle['install']['installationMode'] == 'fresh-install'
assert lifecycle['install']['installedFilesVerified'] == len(manifest['files']) - 4
assert lifecycle['smoke']['smokeConditions'] == 12 and lifecycle['smoke']['executableBytesUnchanged']
assert lifecycle['remove']['payloadRemoved'] and lifecycle['remove']['kernelRestrictionUnchanged']
assert len(lifecycle['remove']['userProbesPreserved']) == 4
for os_name in ('arch', 'ubuntu'):
log = (RESULTS / ('package-tests-' + os_name + '.log')).read_text()
assert 'Ran 8 tests' in log and log.rstrip().endswith('OK')
assert read('shutdown.json')['allGuestsStopped']
record = {**prior, 'schemaVersion': 3, 'recordedAtUtc': datetime.now(timezone.utc).isoformat(),
'scope': 'Notice-only Ubuntu validation3 package; all compiled product sources and executables unchanged '
'from the URL-fix record. Eight packaging tests on Arch and Ubuntu, two identical deb builds, '
'archive verification and 12 X11/Wayland launches plus install/remove/purge on the existing SDK-free VM.',
'ubuntuPackage': package, 'packageLifecycle': lifecycle,
'noticeSupplement': supplement,
'packageChanges': {'added': added, 'removed': removed, 'changed': changed,
'selfManifestAlsoChanged': True, 'allOtherPayloadIdentical': True},
'previousValidation': {'record': 'tests/results/qt-notice-supplement/prior-validation-record.json',
'sha256': sha(RESULTS / 'prior-validation-record.json'),
'scope': 'All 22 CTest groups per OS, native portal cases and performance stay bound '
'to their original runs. Product bytes are unchanged. Portal ran on package '
'validation2; its OS integration was not rerun for notice-only validation3.'},
'packageVmScope': 'Existing dedicated Ubuntu 24.04 VM with no SDK/build tree. Test helpers already '
'installed; this is not a newly created OS. Prior validation1 fresh-OS evidence is historical.',
'validationToolsSha256': {**prior['validationToolsSha256'], **{name: sha(ROOT / name) for name in (
'tools/package_ubuntu.py', 'tools/verify_ubuntu_package.py', 'tests/test_ubuntu_package.py',
'tests/source_archives/collect_qt_sdk_supplement.py', 'tests/source_archives/record_qt_supplement.py')}},
'evidenceSha256': {**prior['evidenceSha256'], **{str(path.relative_to(ROOT)): sha(path)
for path in sorted(RESULTS.rglob('*')) if path.is_file() and path.suffix != '.pyc'
and path != RESULTS / 'record.json'}}, 'goalComplete': False}
encoded = json.dumps(record, ensure_ascii=False, indent=2) + '\n'
(RESULTS / 'record.json').write_text(encoded)
(ROOT / 'docs/validation-record.json').write_text(encoded)
print(json.dumps({'additionalNoticeFiles': 2, 'packageFiles': package['payloadFiles'],
'smokeConditions': 12, 'compiledSourcesUnchanged': True, 'goalComplete': False}))
if __name__ == '__main__':
main()
+50
View File
@@ -0,0 +1,50 @@
#!/usr/bin/env python3
"""Compare the pinned toml++ source archive with installed development inputs."""
import argparse
import hashlib
import json
from pathlib import Path
import tarfile
ROOT = Path(__file__).resolve().parents[2]
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--output', type=Path, required=True)
args = parser.parse_args()
pin = json.loads(Path(__file__).with_name('pins.json').read_text())['tomlplusplus']
archive = ROOT / '.deps/source-archives' / pin['name']
data = archive.read_bytes()
if len(data) > pin['maximumBytes'] or any(hashlib.new(name, data).hexdigest() != expected for name, expected in pin['hashes'].items()):
raise SystemExit('Archive does not match the recipe checksums')
headers, license_text = [], None
with tarfile.open(archive) as stream:
for member in stream:
if not member.isfile():
continue
prefix = 'tomlplusplus-3.4.0/'
if not member.name.startswith(prefix) or '..' in Path(member.name).parts:
raise SystemExit('Unexpected source member path')
relative = member.name[len(prefix):]
if not (relative.startswith('include/toml++/') or relative == 'LICENSE'):
continue
if member.size > 8 * 1024 * 1024:
raise SystemExit('Selected source file exceeds its limit')
content = stream.extractfile(member).read()
if relative == 'LICENSE':
license_text = content
else:
installed = Path('/usr') / relative
headers.append({'path': relative, 'bytes': len(content), 'sha256': hashlib.sha256(content).hexdigest(),
'installedMatches': installed.is_file() and installed.read_bytes() == content})
if license_text is None or len(headers) != 51 or not all(row['installedMatches'] for row in headers):
raise SystemExit('Installed header set differs from the pinned source archive')
record = {'archive': str(archive.relative_to(ROOT)), 'sha256': hashlib.sha256(data).hexdigest(),
'headers': headers, 'allInstalledHeadersMatch': True, 'headerCount': len(headers),
'licenseSha256': hashlib.sha256(license_text).hexdigest(),
'licenseMatchesInstalled': license_text == Path('/usr/share/licenses/tomlplusplus/LICENSE').read_bytes(),
'scope': 'Source archive and current installed headers; not a rebuild or proof of every historical compiler input'}
if not record['licenseMatchesInstalled']:
raise SystemExit('Installed license differs from source archive')
args.output.mkdir(parents=True, exist_ok=False)
(args.output / 'LICENSE').write_bytes(license_text)
(args.output / 'verification.json').write_text(json.dumps(record, indent=2) + '\n')
print('Verified 51 toml++ headers and the source license')