from pathlib import Path
from bs4 import BeautifulSoup
import ast, hashlib, json, re

ROOT = Path(__file__).resolve().parents[4]
LEGACY = ROOT / 'outputs/rita-journey-20261002/legacy'
STAGE = LEGACY / 'staged'
INDEX = STAGE / 'index.html'
EVIDENCE = Path(__file__).resolve().parent
BUILDER = LEGACY / 'build_legacy_review.py'


def sha(text):
    return hashlib.sha256(text.encode('utf-8')).hexdigest()


def assignment(tree, name):
    matches = []
    for node in tree.body:
        if isinstance(node, ast.Assign) and any(isinstance(target, ast.Name) and target.id == name for target in node.targets):
            matches.append(ast.literal_eval(node.value))
    assert len(matches) == 1, f'expected one literal {name}, found {len(matches)}'
    return matches[0]


def target_list(markup):
    soup = BeautifulSoup(markup, 'html.parser')
    return [frame.get('src') for frame in soup.select('iframe')], [el['data-cid'] for el in soup.select('[data-cid]')]


def normalized_markup(markup):
    markup = re.sub(r'(<script data-kl-comment-layout="r1">)[\s\S]*?(</script>)', r'\1__LAYOUT_RUNTIME__\2', markup)
    markup = re.sub(r'(<script id="legacy-embedded-layer-adapter">)[\s\S]*?(</script>)', r'\1__EMBEDDED_ADAPTER__\2', markup)
    return markup


tree = ast.parse(BUILDER.read_text())
adapter = assignment(tree, 'adapter')
layout_changes = assignment(tree, 'layout_changes')
layout_re = re.compile(r'(<script data-kl-comment-layout="r1">)([\s\S]*?)(</script>)')
adapter_re = re.compile(r'<script id="legacy-embedded-layer-adapter">[\s\S]*?</script>')

before = INDEX.read_text()
original_hash = sha(before)

# The just-regenerated file contains an appended duplicate of the critique index.
# The existing source already contains all 267 stable critique targets. Drop only
# the second appended copy and its generated tail, preserving the preceding page.
critique_marker = 'id="legacy-critique-index"'
critique_positions = [m.start() for m in re.finditer(re.escape(critique_marker), before)]
trimmed_duplicate = False
if len(critique_positions) == 2:
    cut = critique_positions[1]
    prefix = before[:cut]
    duplicate = before[cut:]
    prefix_cids = [el['data-cid'] for el in BeautifulSoup(prefix, 'html.parser').select('[data-cid]')]
    duplicate_cids = [el['data-cid'] for el in BeautifulSoup(duplicate, 'html.parser').select('[data-cid]')]
    assert len(prefix_cids) == 267 and len(set(prefix_cids)) == 267
    assert len(duplicate_cids) == 267 and set(duplicate_cids) == set(prefix_cids)
    assert prefix.count('id="legacy-embedded-layer-adapter"') == 1
    before = prefix + '</body></html>'
    trimmed_duplicate = True
elif len(critique_positions) != 1:
    raise AssertionError(f'expected one assembled critique index or one duplicate copy, found {len(critique_positions)}')

prepatch_normalized = normalized_markup(before)
frames_before, cids_before = target_list(before)
assert len(frames_before) == 29 and len(set(frames_before)) == 29
assert len(cids_before) == 267 and len(set(cids_before)) == 267

layout_matches = list(layout_re.finditer(before))
assert len(layout_matches) == 1, f'expected one toolbar layout runtime, found {len(layout_matches)}'
layout = layout_matches[0].group(2)
for old, new in layout_changes:
    if old in layout:
        assert layout.count(old) == 1
        layout = layout.replace(old, new, 1)
    else:
        assert layout.count(new) == 1, f'layout is neither old nor patched for: {old}'
before = before[:layout_matches[0].start(2)] + layout + before[layout_matches[0].end(2):]

adapter_matches = list(adapter_re.finditer(before))
assert len(adapter_matches) == 1, f'expected one outer adapter to preserve and replace, found {len(adapter_matches)}'
before = adapter_re.sub(lambda _match: adapter, before, count=1)
assert before.count('id="legacy-embedded-layer-adapter"') == 1
assert 'position:fixed!important;top:8px!important;right:10px!important' in before
assert 'position:relative!important' not in layout
assert 'data-legacy-comment-master' in before

frames_after, cids_after = target_list(before)
assert frames_after == frames_before
assert cids_after == cids_before
assert normalized_markup(before) == prepatch_normalized, 'bytes outside the toolbar layout and adapter changed'

INDEX.write_text(before)

child_baseline = json.loads((EVIDENCE / 'child-files-before.json').read_text())
child_results = {}
for relative, expected in child_baseline['children'].items():
    raw = (STAGE / relative).read_bytes()
    actual = hashlib.sha256(raw).hexdigest()
    child_results[relative] = {'sha256': actual, 'bytes': len(raw), 'unchanged': actual == expected['sha256']}
assert all(row['unchanged'] for row in child_results.values())

proof = {
    'status': 'PASS',
    'inputIndexSha256': original_hash,
    'outputIndexSha256': sha(before),
    'trimmedBuilderDuplicateCritiqueCopy': trimmed_duplicate,
    'nonRuntimeMarkupPreservedAfterTrim': True,
    'layoutAndAdapterOnlyChanged': True,
    'masterId': 'komplex-rita-funnel-20260926-master',
    'adapterScriptCount': before.count('id="legacy-embedded-layer-adapter"'),
    'toolbarRule': 'fixed at top 8px and right 10px in the viewport',
    'iframeCount': len(frames_after),
    'iframeSourcesPreserved': True,
    'critiqueTargets': len(cids_after),
    'critiqueTargetsUnique': len(set(cids_after)) == len(cids_after),
    'childFiles': child_results,
}
(EVIDENCE / 'source-repair-proof.json').write_text(json.dumps(proof, ensure_ascii=False, indent=2))
print(json.dumps({k:v for k,v in proof.items() if k != 'childFiles'}, ensure_ascii=False, indent=2))
print('identical child files:', sum(row['unchanged'] for row in child_results.values()), '/', len(child_results))
