#!/usr/bin/env python3
"""Retrieve only five pinned public text references for Labs 08/09. Never import them."""
import argparse
import datetime
import hashlib
import json
from pathlib import Path
import urllib.request

FILES = {
    'gpt2-config.json': ('https://huggingface.co/openai-community/gpt2/resolve/7979bb77dc27f20ed20949aa4b91db714c64751f/config.json', 665, '0daed7749b4f02b8f76240d5444551d7b08712dab4d0adb8239c56ba823bb7b4'),
    'gemma-config.py': ('https://raw.githubusercontent.com/google/gemma_pytorch/cb7c0152a369e43908e769eb09e1ce6043afe084/gemma/config.py', 10856, '43bcc8e024815e18e41045fd168509669ffc2af506061fa598db795e021dafd1'),
    'gpt2-model.py': ('https://raw.githubusercontent.com/openai/gpt-2/c2dae27c1029770cea40978813f17a5fd545b883/src/model.py', 6503, '2ff5065f5cac3dc93065f8a3f36eb013e407972a80abcd4ee6b89a7138a0d5a6'),
    'gemma-model.py': ('https://raw.githubusercontent.com/google/gemma_pytorch/cb7c0152a369e43908e769eb09e1ce6043afe084/gemma/model.py', 28224, 'e4dd8b696efbb36a1dd7ba8a2cc0d206060275f9784f7470a65d0d096507c3f5'),
    'gpt-oss-config.json': ('https://huggingface.co/openai/gpt-oss-20b/resolve/f81fef1ddd90d214968e951a76834f1ded130a18/config.json', 1806, '3a2a26ded679375b7928ddeca59764df7cea83220c1961035f6d6e232659e9ce'),
}


def retrieve(destination, offline=False):
    destination = Path(destination)
    if destination.is_symlink():
        raise ValueError('Dedicated source directory must not be a symlink')
    destination.mkdir(parents=True, exist_ok=True)
    if set(p.name for p in destination.iterdir()) - set(FILES) - {'provenance.json'}:
        raise ValueError('Dedicated directory contains unexpected files')
    rows = []
    for name, (url, size, digest) in FILES.items():
        target = destination / name
        if target.is_symlink():
            raise ValueError('Reference must not be a symlink')
        if target.exists():
            data = target.read_bytes()
        elif offline:
            raise ValueError(f'Missing offline reference: {name}')
        else:
            with urllib.request.urlopen(url, timeout=30) as response:
                data = response.read(size + 1)
        if len(data) != size or hashlib.sha256(data).hexdigest() != digest:
            raise ValueError(f'Size/SHA-256 mismatch: {name}')
        data.decode('utf-8')
        if not target.exists():
            target.write_bytes(data)
        rows.append({'file': name, 'url': url, 'bytes': size, 'sha256': digest})
    report = {'kind': 'public reference text only; no execution or model loading',
              'verified_at_utc': datetime.datetime.now(datetime.timezone.utc).isoformat(),
              'files': rows, 'notices': 'Original downloaded bytes and embedded notices retained unchanged. Consult the pinned publisher repositories for their full licenses and terms.'}
    provenance = destination / 'provenance.json'
    if provenance.is_symlink():
        raise ValueError('Provenance must not be a symlink')
    if not provenance.exists():
        provenance.write_text(json.dumps(report, indent=2) + '\n', encoding='utf-8')
    return report


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument('--output', type=Path, default=Path('config-references'))
    parser.add_argument('--offline', action='store_true')
    parser.add_argument('--plan', action='store_true')
    args = parser.parse_args()
    print(json.dumps({'files': FILES, 'total_bytes': sum(v[1] for v in FILES.values())} if args.plan else retrieve(args.output, args.offline), indent=2))


if __name__ == '__main__':
    main()
