#!/usr/bin/env python3
"""Read-only integrity verifier for the postdeployment screenshot archive."""
from __future__ import annotations

import hashlib
import io
import json
import sys
import zipfile
from collections import Counter
from pathlib import Path

from PIL import Image

ROOT = Path("/Users/agency/Documents/Agty/KomplexLogopédia/outputs/rita-journey-20261002")
POST = ROOT / "evidence/inspection-post/inspection-manifest.json"
LEGACY = ROOT / "evidence/luna-post-a-inputs/legacy-supplement.json"
PUBLIC_POST = ROOT / "site/review/screenshots/post/capture-index.json"
PUBLIC_LEGACY = ROOT / "site/review/screenshots/post/legacy-supplement-index.json"
OUT = ROOT / "evidence/archive-integrity-final/receipt.json"


def sha(data: bytes) -> str:
    return hashlib.sha256(data).hexdigest()


def read_json(path: Path):
    return json.loads(path.read_text(encoding="utf-8"))


def png_dimensions(data: bytes) -> tuple[int, int]:
    # verify() parses all PNG chunks; reopen/load() validates decompressed pixels.
    with Image.open(io.BytesIO(data)) as im:
        if im.format != "PNG":
            raise ValueError(f"expected PNG, got {im.format}")
        size = im.size
        im.verify()
    with Image.open(io.BytesIO(data)) as im:
        im.load()
        return size


def public_equivalent(private: dict, public: dict) -> bool:
    def redact(value):
        if isinstance(value, list):
            return [redact(item) for item in value]
        if isinstance(value, dict):
            return {k: redact(v) for k, v in value.items() if k != "file"}
        return value
    # Sheet assembly references and private absolute source paths are intentionally not public.
    private_core = {k: redact(v) for k, v in private.items() if k != "sheets"}
    return private_core == public


def main() -> int:
    post, legacy = read_json(POST), read_json(LEGACY)
    public_post, public_legacy = read_json(PUBLIC_POST), read_json(PUBLIC_LEGACY)
    errors: list[str] = []
    full_page_errors: list[str] = []
    counts = Counter()
    decoded_hashes: set[str] = set()
    source_cache: dict[str, bytes] = {}
    archive_cache: dict[str, zipfile.ZipFile] = {}
    archive_entries: dict[str, set[str]] = {}
    archive_capture: dict[str, dict | None] = {}

    # Published JSON retains every substantive record field, with only private absolute source paths omitted.
    if not public_equivalent(post, public_post):
        errors.append("public capture-index.json is not the private manifest with file paths omitted")
    else:
        counts["public_post_identity"] = len(post["records"])
    if not public_equivalent(legacy, public_legacy):
        errors.append("public legacy-supplement-index.json is not the private supplement with file paths omitted")
    else:
        counts["public_legacy_identity"] = len(legacy["records"])

    # Every declared post archive is present, byte-hashed, structurally valid, and has exactly its declared entries.
    if len(post["archives"]) != 791:
        errors.append(f"expected 791 archive declarations, found {len(post['archives'])}")
    for declared in post["archives"]:
        rel = declared["path"]
        path = ROOT / "site" / rel
        try:
            raw = path.read_bytes()
            if len(raw) != declared["bytes"]:
                errors.append(f"archive byte size mismatch: {rel}")
            if sha(raw) != declared["sha256"]:
                errors.append(f"archive SHA-256 mismatch: {rel}")
            zf = zipfile.ZipFile(io.BytesIO(raw))
            bad = zf.testzip()
            if bad:
                errors.append(f"corrupt ZIP member {bad}: {rel}")
            actual_entries = set(zf.namelist())
            expected_entries = set(declared["entries"])
            if actual_entries != expected_entries or len(zf.namelist()) != len(actual_entries):
                errors.append(f"archive entry list mismatch: {rel}")
            archive_cache[rel] = zf
            archive_entries[rel] = actual_entries
            try:
                archive_capture[rel] = json.loads(zf.read("capture.json"))
            except KeyError:
                archive_capture[rel] = None
            counts["archives_verified"] += 1
        except Exception as exc:
            errors.append(f"archive unreadable {rel}: {type(exc).__name__}: {exc}")

    # Full-page references are deliberately separate from records. Verify every one independently.
    full_page_decoded_hashes: set[str] = set()
    if len(post["full_page_references"]) != 416:
        full_page_errors.append(f"expected 416 full-page references, found {len(post['full_page_references'])}")
    for ref in post["full_page_references"]:
        counts["full_page_references_checked"] += 1
        try:
            source_path = Path(ref["file"])
            source_raw = source_path.read_bytes()
            source_hash = sha(source_raw)
            if source_hash not in full_page_decoded_hashes:
                dimensions = png_dimensions(source_raw)
                full_page_decoded_hashes.add(source_hash)
            else:
                with Image.open(io.BytesIO(source_raw)) as im:
                    dimensions = im.size
            vw, vh = ref["viewport"]["width"], ref["viewport"]["height"]
            if dimensions[0] != vw or dimensions[1] < vh:
                full_page_errors.append(f"full-page dimensions incompatible with recorded viewport: {ref['group']}")
            zf = archive_cache.get(ref["archive_path"])
            if zf is None or ref["archive_entry"] not in archive_entries.get(ref["archive_path"], set()):
                full_page_errors.append(f"missing full-page archive entry: {ref['archive_path']}!{ref['archive_entry']}")
                continue
            archived = zf.read(ref["archive_entry"])
            if archived != source_raw or sha(archived) != source_hash:
                full_page_errors.append(f"full-page archive bytes differ from source: {ref['group']}")
            capture = archive_capture.get(ref["archive_path"])
            if capture is None:
                full_page_errors.append(f"full-page archive has no capture.json context: {ref['group']}")
                continue
            item = capture.get("item", {})
            requested = capture.get("viewport", {}).get("requested", {})
            if item.get("id") != ref["asset_id"]:
                full_page_errors.append(f"full-page asset ID mismatch: {ref['group']}")
            if capture.get("url") != ref["url"]:
                full_page_errors.append(f"full-page URL mismatch: {ref['group']}")
            if requested != ref["viewport"]:
                full_page_errors.append(f"full-page viewport mismatch: {ref['group']}")
            if capture.get("finishedAt") != ref["captured_at"]:
                full_page_errors.append(f"full-page time mismatch: {ref['group']}")
            if not any(x.get("type") == "full-page" and x.get("filename") == ref["archive_entry"] for x in capture.get("captures", [])):
                full_page_errors.append(f"full-page capture.json entry mismatch: {ref['group']}")
            document = capture.get("basis", {}).get("document", {})
            if document and dimensions != (document.get("width"), document.get("height")):
                full_page_errors.append(f"full-page dimensions mismatch capture.json document: {ref['group']}")
            counts["full_page_reference_entry_resolutions"] += 1
        except Exception as exc:
            full_page_errors.append(f"full-page reference unreadable {ref.get('group')}: {type(exc).__name__}: {exc}")
    errors.extend(full_page_errors)

    # The legacy supplement deliberately adds a sixth archive outside the 791 post-manifest ZIPs.
    outer_zip_rel = "review/screenshots/post/legacy-outer-marker/all-six-originals.zip"
    outer_zip = ROOT / "site" / outer_zip_rel
    try:
        outer_raw = outer_zip.read_bytes()
        outer_zf = zipfile.ZipFile(io.BytesIO(outer_raw))
        bad = outer_zf.testzip()
        if bad:
            errors.append(f"corrupt outer ZIP member {bad}")
        archive_cache[outer_zip_rel] = outer_zf
        archive_entries[outer_zip_rel] = set(outer_zf.namelist())
        archive_capture[outer_zip_rel] = None
        counts["supplemental_outer_archive_verified"] = 1
    except Exception as exc:
        errors.append(f"outer six-originals ZIP unreadable: {type(exc).__name__}: {exc}")

    # Legacy archive records use the preserved public capture manifest rather than capture.json inside their ZIPs.
    legacy_capture_records: dict[str, dict] = {}
    for manifest_path in (ROOT / "site/review/screenshots/post").glob("legacy-*-capture-manifest.json"):
        data = read_json(manifest_path)
        for item in data.get("records", []):
            legacy_capture_records[item["name"]] = item
    outer_data = read_json(ROOT / "site/review/screenshots/post/legacy-outer-marker/capture-manifest.json")
    for item in outer_data.get("captures", []):
        legacy_capture_records[item["name"]] = item

    def validate_metadata(rec: dict, archive_metadata: dict | None) -> None:
        """Check all metadata fields that have an independently retained archive/public counterpart."""
        name = rec["name"]
        if archive_metadata is not None:
            if archive_metadata.get("url") != rec["url"]:
                errors.append(f"capture.json URL mismatch: {name}")
            item = archive_metadata.get("item", {})
            if item.get("id") != rec["asset_id"]:
                errors.append(f"capture.json asset ID mismatch: {name}")
            requested = archive_metadata.get("viewport", {}).get("requested", {})
            if requested != rec["viewport"]:
                errors.append(f"capture.json viewport mismatch: {name}")
            cap = next((x for x in archive_metadata.get("captures", []) if x.get("filename") == name), None)
            if cap is None:
                errors.append(f"capture.json missing capture metadata: {name}")
                return
            pairs = [("type", "type"), ("sha256", "sha256"), ("capturedAt", "captured_at")]
            for a, b in pairs:
                if a in cap and b in rec and cap[a] != rec[b]:
                    errors.append(f"capture.json {a} mismatch: {name}")
            if "pixels" in cap and rec.get("pixels") != cap["pixels"]:
                errors.append(f"capture.json pixels mismatch: {name}")
            if "actualScroll" in cap and rec.get("actualScroll") != cap["actualScroll"]:
                errors.append(f"capture.json actualScroll mismatch: {name}")
            counts["capture_json_identity"] += 1
        else:
            cap = legacy_capture_records.get(name)
            if cap is None:
                # The published post/supplement manifest retains this record's metadata.
                counts["published_manifest_metadata_identity"] += 1
                return
            if cap.get("url", cap.get("basis", {}).get("href")) != rec["url"]:
                errors.append(f"legacy public URL mismatch: {name}")
            recorded_viewport = cap.get("requestedViewport", cap.get("basis", {}).get("viewport"))
            if recorded_viewport is not None:
                recorded_viewport = {k: v for k, v in recorded_viewport.items() if k in ("width", "height", "dpr")}
                expected_viewport = {k: v for k, v in rec["viewport"].items() if k in recorded_viewport}
                if recorded_viewport != expected_viewport:
                    errors.append(f"legacy public viewport mismatch: {name}")
            if cap.get("state") != rec.get("state"):
                errors.append(f"legacy public state mismatch: {name}")
            if cap.get("sha256") != rec.get("sha256"):
                errors.append(f"legacy public SHA mismatch: {name}")
            captured = cap.get("capturedAt", cap.get("captured_at"))
            if captured != rec.get("captured_at"):
                errors.append(f"legacy public time mismatch: {name}")
            if cap.get("id") and cap.get("id") != rec["asset_id"]:
                errors.append(f"legacy public asset ID mismatch: {name}")
            counts["legacy_public_metadata_identity"] += 1

    all_records = [("post", rec) for rec in post["records"]] + [("legacy", rec) for rec in legacy["records"]]
    for source, rec in all_records:
        counts[f"{source}_records_checked"] += 1
        # Excluded initial-capture *directories* must never be accepted. Filename state 'restored-initial' is valid.
        parts = Path(rec["file"]).parts
        forbidden = [p for p in parts if p.startswith("legacy-capture-artifacts-initial-") or p == "capture-internal-initial-failures"]
        if forbidden:
            errors.append(f"accepted excluded initial capture directory {forbidden[0]}: {rec['name']}")
        try:
            path = Path(rec["file"])
            raw = source_cache.setdefault(str(path), path.read_bytes())
            actual_sha = sha(raw)
            if rec.get("sha256") and actual_sha != rec["sha256"]:
                errors.append(f"source SHA-256 mismatch: {rec['name']}")
            dimensions = None
            decode_key = actual_sha
            if decode_key not in decoded_hashes:
                dimensions = png_dimensions(raw)
                decoded_hashes.add(decode_key)
            # dimensions are always checked even if decoding work was deduplicated.
            if dimensions is None:
                with Image.open(io.BytesIO(raw)) as im:
                    dimensions = im.size
            expected = rec.get("pixels")
            if expected and dimensions != (expected["width"], expected["height"]):
                errors.append(f"recorded PNG dimensions mismatch: {rec['name']}")
            elif not expected:
                vw, vh = rec["viewport"]["width"], rec["viewport"]["height"]
                if rec.get("type") == "full-page":
                    if dimensions[0] != vw or dimensions[1] < vh:
                        errors.append(f"full-page PNG dimensions incompatible with viewport: {rec['name']}")
                elif dimensions != (vw, vh):
                    errors.append(f"viewport PNG dimensions mismatch: {rec['name']}")
            rel = rec["archive_path"]
            zf = archive_cache.get(rel)
            if zf is None or rec["archive_entry"] not in archive_entries.get(rel, set()):
                errors.append(f"missing archive entry: {rel}!{rec['archive_entry']}")
            else:
                archived = zf.read(rec["archive_entry"])
                if archived != raw:
                    errors.append(f"archive bytes differ from source: {rel}!{rec['archive_entry']}")
                if sha(archived) != actual_sha:
                    errors.append(f"archive extracted SHA mismatch: {rel}!{rec['archive_entry']}")
                validate_metadata(rec, archive_capture.get(rel))
                counts["archive_entries_resolved"] += 1
        except Exception as exc:
            errors.append(f"record unreadable {rec.get('name')}: {type(exc).__name__}: {exc}")

    # Explicit six originals: direct public PNG and archive member must both equal the supplement's original bytes.
    outer = [r for r in legacy["records"] if r.get("archive_path") == "review/screenshots/post/legacy-outer-marker/all-six-originals.zip"]
    if len(outer) != 6:
        errors.append(f"expected six outer-marker records, found {len(outer)}")
    expected_outer_entries = {rec["archive_entry"] for rec in outer} | {"capture-manifest.json"}
    if archive_entries.get(outer_zip_rel, set()) != expected_outer_entries:
        errors.append("all-six-originals.zip entry list does not exactly match the six declared originals")
    else:
        try:
            zipped_manifest = archive_cache[outer_zip_rel].read("capture-manifest.json")
            public_manifest = (ROOT / "site/review/screenshots/post/legacy-outer-marker/capture-manifest.json").read_bytes()
            if zipped_manifest != public_manifest:
                errors.append("all-six-originals.zip capture-manifest.json differs from the public capture manifest")
            else:
                counts["outer_capture_manifest_verified"] = 1
        except Exception as exc:
            errors.append(f"outer ZIP capture manifest unreadable: {type(exc).__name__}: {exc}")
    for rec in outer:
        direct = ROOT / "site" / rec["direct_path"]
        try:
            direct_raw = direct.read_bytes()
            if direct_raw != source_cache[str(Path(rec["file"]))]:
                errors.append(f"direct outer-marker PNG differs from source: {rec['name']}")
            if sha(direct_raw) != rec["sha256"]:
                errors.append(f"direct outer-marker PNG SHA mismatch: {rec['name']}")
            png_dimensions(direct_raw)
            counts["outer_direct_pngs_verified"] += 1
        except Exception as exc:
            errors.append(f"direct outer-marker PNG unreadable {rec['name']}: {type(exc).__name__}: {exc}")

    receipt = {
        "status": "PASS" if not errors else "FAIL",
        "post_records_declared": len(post["records"]),
        "post_records_checked": counts["post_records_checked"],
        "post_full_page_references_declared": len(post["full_page_references"]),
        "full_page_references_checked": counts["full_page_references_checked"],
        "full_page_reference_entry_resolutions": counts["full_page_reference_entry_resolutions"],
        "unique_full_page_hashes_decoded": len(full_page_decoded_hashes),
        "full_page_reference_errors": full_page_errors,
        "full_page_reference_error_count": len(full_page_errors),
        "archives_declared": len(post["archives"]),
        "archives_verified": counts["archives_verified"],
        "archive_entries_resolved": counts["archive_entries_resolved"],
        "unique_source_byte_hashes_decoded": len(decoded_hashes),
        "legacy_records_declared": len(legacy["records"]),
        "legacy_records_checked": counts["legacy_records_checked"],
        "legacy_asset_count": legacy["legacy_asset_count"],
        "legacy_expanded_card_context_count": legacy["expanded_card_context_count"],
        "legacy_outer_marker_context_count": legacy["outer_marker_context_count"],
        "outer_direct_pngs_verified": counts["outer_direct_pngs_verified"],
        "public_post_identity_records": counts["public_post_identity"],
        "public_legacy_identity_records": counts["public_legacy_identity"],
        "capture_json_metadata_records": counts["capture_json_identity"],
        "legacy_public_metadata_records": counts["legacy_public_metadata_identity"],
        "errors": errors,
    }
    OUT.write_text(json.dumps(receipt, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
    print(json.dumps(receipt, indent=2, ensure_ascii=False))
    return 0 if not errors else 1


if __name__ == "__main__":
    raise SystemExit(main())
