Files
2026-09-21 13:41:40 +09:00

99 lines
5.0 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Collect pinned public source archives; never execute downloaded code."""
import argparse
from concurrent.futures import ThreadPoolExecutor
from datetime import datetime, timezone
import hashlib
import json
from pathlib import Path
import time
import urllib.parse
import urllib.request
ROOT = Path(__file__).resolve().parents[2]
PINS = Path(__file__).with_name('pins.json')
HOSTS = {'download.qt.io', 'ftp.jaist.ac.jp', 'github.com', 'codeload.github.com'}
class Redirects(urllib.request.HTTPRedirectHandler):
def redirect_request(self, req, fp, code, msg, headers, newurl):
parsed = urllib.parse.urlparse(newurl)
if parsed.scheme != 'https' or parsed.hostname not in HOSTS:
raise ValueError('Source archive redirect left the pinned HTTPS providers')
return super().redirect_request(req, fp, code, msg, headers, newurl)
def fetch(name, pin, output, cache):
result = {'name': name, 'pin': pin, 'success': False,
'retrievedAtUtc': datetime.now(timezone.utc).isoformat()}
target = cache / pin['name']; partial = target.with_name(target.name + '.part')
before = time.monotonic()
try:
if target.exists() or partial.exists():
raise ValueError('Refusing to replace an existing archive or partial download')
opener = urllib.request.build_opener(Redirects())
if 'metadataUrl' in pin:
with opener.open(pin['metadataUrl'], timeout=30) as source:
metadata = source.read(128 * 1024 + 1)
if len(metadata) > 128 * 1024:
raise ValueError('Metadata exceeded its finite limit')
if not all(value.encode() in metadata for value in pin['hashes'].values()):
raise ValueError('Published Qt checksum differs from the pin')
result['metadataSha256'] = hashlib.sha256(metadata).hexdigest()
# Mirror pages may contain the requesting IP. Keep the digest and
# matched public fields rather than copying that incidental data.
result['publishedHashesMatched'] = True
else:
metadata = (ROOT / pin['metadataFile']).read_bytes()
if not all(value.encode() in metadata for value in pin['hashes'].values()):
raise ValueError('Stored recipe differs from the pinned checksums')
result['metadataSha256'] = hashlib.sha256(metadata).hexdigest()
limit = pin.get('bytes', pin.get('maximumBytes'))
hashes = {algorithm: hashlib.new(algorithm) for algorithm in {*pin['hashes'], 'sha256'}}
count, last = 0, before
with opener.open(pin['url'], timeout=30) as source, partial.open('xb') as file:
result['resolvedUrl'] = source.url
while block := source.read(1024 * 1024):
count += len(block)
if count > limit or time.monotonic() - before > 900:
raise ValueError('Download exceeded size or time limit')
file.write(block)
for digest in hashes.values():
digest.update(block)
if time.monotonic() - last > 10:
print(f'{name}: {count / 1048576:.1f} MiB received', flush=True)
last = time.monotonic()
result.update(bytes=count, hashes={algorithm: digest.hexdigest() for algorithm, digest in hashes.items()})
if pin.get('bytes', count) != count or any(result['hashes'][algorithm] != digest for algorithm, digest in pin['hashes'].items()):
raise ValueError('Archive size or checksum mismatch; partial retained as unverified')
partial.rename(target)
result.update(success=True, archive=str(target.relative_to(ROOT)))
except Exception as error:
result['error'] = str(error)
result['elapsedSeconds'] = time.monotonic() - before
(output / (name + '.json')).write_text(json.dumps(result, indent=2) + '\n')
print(json.dumps({'name': name, 'success': result['success'], 'error': result.get('error')}), flush=True)
return result['success']
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--output', type=Path, required=True)
parser.add_argument('--cache', type=Path, required=True)
parser.add_argument('--artifact', action='append', choices=['qt-everywhere', 'tomlplusplus'])
args = parser.parse_args()
output, cache = args.output.resolve(), args.cache.resolve()
if not output.is_relative_to(ROOT) or not cache.is_relative_to(ROOT):
parser.error('Keep this task’s evidence and archives inside the workspace')
output.mkdir(parents=True, exist_ok=False); cache.mkdir(parents=True, exist_ok=True)
pins = json.loads(PINS.read_text())
names = args.artifact or list(pins)
with ThreadPoolExecutor(max_workers=2) as executor:
futures = [executor.submit(fetch, name, pins[name], output, cache) for name in names]
successes = [future.result() for future in futures]
return 0 if all(successes) else 1
if __name__ == '__main__':
raise SystemExit(main())