#!/usr/bin/env python3
"""Fetch OECD documentation and its 2022 ICIO table archived by the World Bank.

The official World Bank catalog identifies the archived table as the regular
2025 edition / January 2026 update; the direct OECD ZIP was HTTP-403 blocked.
No authentication or credentials are used. Raw files live
outside the application repository. Existing files are hash-verified when a
manifest exists; downloads are atomic and network failures leave .part files.
"""
import argparse
import csv
from datetime import datetime, timezone
import hashlib
import io
import json
from pathlib import Path
import shutil
from urllib.request import Request, urlopen
import zipfile

BASE = "https://webfs-sti.oecd.org/files/STI-PIE/ICIO/2025/"
SOURCES = {
    "ReadMe_ICIO_small.xlsx": BASE + "ReadMe_ICIO_small.xlsx",
    "ICIO2025annex.pdf": BASE + "ICIO2025annex.pdf",
    "worldbank-catalog-511.html": "https://reproducibility.worldbank.org/catalog/511/study-description",
    "worldbank-package-README.pdf": "https://reproducibility.worldbank.org/catalog/511/download/1497",
    "RR_WLD_2026_593.zip": "https://reproducibility.worldbank.org/catalog/511/download/1471",
}


def sha256(path):
    digest = hashlib.sha256()
    with path.open("rb") as stream:
        for chunk in iter(lambda: stream.read(1024 * 1024), b""):
            digest.update(chunk)
    return digest.hexdigest()


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--raw-dir", type=Path, required=True)
    parser.add_argument("--manifest", type=Path, default=Path(__file__).with_name("source-manifest.json"))
    parser.add_argument("--offline", action="store_true", help="Verify manifest and cached source hashes without network")
    args = parser.parse_args()
    args.raw_dir.mkdir(parents=True, exist_ok=True)
    prior = json.loads(args.manifest.read_text()) if args.manifest.exists() else {"sources": []}
    known = {item["filename"]: item for item in prior["sources"]}
    manifest = {
        "edition": "2025 regular ICIO", "revision": "mid-January 2026 second revision", "observation_year": 2022,
        "data_producer": "OECD", "landing_page": "https://www.oecd.org/en/data/datasets/inter-country-input-output-tables.html",
        "primary_archive_url": BASE + "2016-2022_SML.zip",
        "primary_archive_retrieval": {"status": "HTTP 403 / Cloudflare challenge", "attempt_date_utc": "2026-09-24", "sha256": None},
        "retrieval_archive": "World Bank Reproducible Research Repository, catalog 511 / RR_WLD_2026_593 / DOI 10.60572/00tk-h866",
        "archive_vintage_evidence": "World Bank catalog Datasets / ICIO identifies 2025 edition, January 2026 update of regular ICIO; year 2022; data_raw/2022_SML.csv; accessed March 2026.",
        "provenance_limit": "The table is the OECD file preserved in this documented official World Bank archive. Byte identity to the currently inaccessible direct OECD ZIP has not been independently established.",
        "sources": [],
    }
    for filename, url in SOURCES.items():
        target = args.raw_dir / filename
        if filename in known and target.exists():
            if sha256(target) != known[filename]["sha256"]:
                raise ValueError("Cached file checksum changed: " + filename)
            entry = known[filename]
        else:
            if args.offline:
                raise FileNotFoundError("Verified offline source unavailable: " + filename)
            started = datetime.now(timezone.utc).isoformat()
            request = Request(url, headers={"User-Agent": "Mozilla/5.0"})
            partial = target.with_suffix(target.suffix + ".part")
            print("Downloading " + filename, flush=True)
            with urlopen(request, timeout=60) as response, partial.open("wb") as output:
                headers = {name: response.headers.get(name) for name in ("Content-Type", "Content-Length", "Last-Modified", "ETag")}
                final_url = response.geturl()
                shutil.copyfileobj(response, output, length=1024 * 1024)
            partial.replace(target)
            entry = {"filename": filename, "source_url": url, "final_url": final_url, "retrieval_started_utc": started, "retrieved_at_utc": datetime.now(timezone.utc).isoformat(), "bytes": target.stat().st_size, "sha256": sha256(target), "http_metadata": headers}
            if filename in known and entry["sha256"] != known[filename]["sha256"]:
                if filename != "worldbank-catalog-511.html":
                    raise ValueError("Downloaded source changed from frozen checksum: " + filename)
                # Catalog pages contain download/view counters. A refreshed
                # evidence page is recorded, never passed off as frozen bytes.
                entry["previous_evidence_snapshot_sha256"] = known[filename]["sha256"]
                entry["metadata_refresh_note"] = "Dynamic catalog HTML changed; dataset/archive hashes remain frozen and must still match."
        manifest["sources"].append(entry)
        print(filename + ": " + entry["sha256"], flush=True)
    archive_path = args.raw_dir / "RR_WLD_2026_593.zip"
    with zipfile.ZipFile(archive_path) as archive:
        members = archive.infolist()
        candidates = [item for item in members if Path(item.filename).name == "2022_SML.csv"]
        if len(candidates) != 1:
            raise ValueError("Expected exactly one 2022 CSV in source ZIP: " + repr([item.filename for item in members]))
        item = candidates[0]
        target = args.raw_dir / Path(item.filename).name
        if not target.exists():
            with archive.open(item) as source, target.open("wb") as output:
                shutil.copyfileobj(source, output, length=1024 * 1024)
        extracted = {"filename": target.name, "archive_member": item.filename, "archive_sha256": sha256(archive_path), "bytes": target.stat().st_size, "sha256": sha256(target), "zip_crc32": format(item.CRC, "08x"), "archive_member_timestamp": list(item.date_time)}
        if "extracted_table" in prior and extracted != prior["extracted_table"]:
            raise ValueError("Extracted table changed from source manifest")
        manifest["extracted_table"] = extracted
        hash_members = [i for i in members if Path(i.filename).name == "data_hash_report.csv"]
        if len(hash_members) != 1:
            raise ValueError("Expected one World Bank package data hash report")
        hash_bytes = archive.read(hash_members[0])
        hash_rows = list(csv.DictReader(io.StringIO(hash_bytes.decode("utf-8-sig"))))
        table_hash_row = next(row for row in hash_rows if row["filename"] == target.name)
        if table_hash_row["sha256sum"] != extracted["sha256"]:
            raise ValueError("Extracted table does not match publisher package hash report")
        hash_target = args.raw_dir / "data_hash_report.csv"
        if not hash_target.exists():
            hash_target.write_bytes(hash_bytes)
        elif hash_target.read_bytes() != hash_bytes:
            raise ValueError("Cached package hash report differs from archive")
        manifest["publisher_hash_verification"] = {"archive_member": hash_members[0].filename, "hash_report_sha256": hashlib.sha256(hash_bytes).hexdigest(), "table_hash_row": table_hash_row, "matches_extracted_table": True}
        manifest["archive_members"] = [{"filename": i.filename, "bytes": i.file_size, "compressed_bytes": i.compress_size} for i in members]
    pending_manifest = args.manifest.with_suffix(args.manifest.suffix + ".part")
    pending_manifest.write_text(json.dumps(manifest, indent=2) + "\n")
    pending_manifest.replace(args.manifest)
    print("Ready: " + str(target), flush=True)


if __name__ == "__main__":
    main()
