#!/usr/bin/env python3
"""Reproduce the fixed-cohort findings and check the delivered Markdown/CSVs.

Run: python audit-recompute.py [folder]
Standard library only. This verifies arithmetic and artifact consistency, not
live laws, individual eligibility, website behavior or current source changes.
"""
from __future__ import annotations
import csv, io, json, re, sys
from collections import Counter, defaultdict
from pathlib import Path

ROOT = Path(sys.argv[1]).resolve() if len(sys.argv)>1 else Path(__file__).resolve().parent
SUFFIX = '2026-09-29.csv'
checks = []
def check(label, condition):
    checks.append({'check':label, 'passed':bool(condition)})
    if not condition:
        raise AssertionError(label)
def load(name):
    with (ROOT/name).open(encoding='utf-8',newline='') as f:
        return list(csv.DictReader(f))
R = 'electrical-license-reciprocity-routes-'+SUFFIX
S = 'electrical-license-reciprocity-by-state-'+SUFFIX
C = 'electrical-license-reciprocity-class-comparison-'+SUFFIX
routes, states, paired = load(R), load(S), load(C)
journey = [r for r in routes if r['class_level']=='Journeyman' and r['counted_in_analysis']=='Yes']
master = [r for r in routes if r['class_level']=='Master/contractor' and r['counted_in_analysis']=='Yes']
jsets, msets = defaultdict(set), defaultdict(set)
for r in journey:jsets[r['destination_code']].add(r['origin_code'])
for r in master:msets[r['destination_code']].add(r['origin_code'])
check('237 source-list records', len(routes)==237)
check('Unique stable route IDs', len({r['route_id'] for r in routes})==len(routes))
check('175 journey records across 17 selected lists', len(journey)==175 and len(jsets)==17)
check('No duplicated journey origin/destination pairs',sum(map(len,jsets.values()))==len(journey))
check('54 counted master/contractor/supervising entries', len(master)==54 and sum(map(len,msets.values()))==54)
check('51 unique directory jurisdictions',len(states)==51 and len({r['code'] for r in states})==51)
check('11 paired boards',len(paired)==11 and set(msets)=={r['code'] for r in paired})
for r in paired:
    check('Paired counts for '+r['state'],int(r['journeyman_list_states'])==len(jsets[r['code']]) and int(r['master_or_contractor_list_states'])==len(msets[r['code']]))
paired_j=sum(len(jsets[r['code']]) for r in paired)
check('Paired denominator = 110; ratio rounds to 49.1%',paired_j==110 and round(100*len(master)/paired_j,1)==49.1)
check('NH current selected table = 11 journey, 10 master',len(jsets['NH'])==11 and len(msets['NH'])==10)
frequency=Counter(r['origin_code'] for r in journey)
check('27 represented origins; 24 zero-count jurisdictions',len(frequency)==27 and 51-len(frequency)==24)
check('Nine origins appear on one or two lists',sum(n in (1,2) for n in frequency.values())==9)
check('AR, MT and WY tie at 12',max(frequency.values())==12 and {k for k,v in frequency.items() if v==12}=={'AR','MT','WY'})
for r in states:
    check('Directory origin frequency: '+r['code'],int(r['times_named_on_17_journeyman_lists'])==frequency[r['code']])
edges={(r['destination_code'],r['origin_code']) for r in journey if r['origin_code'] in jsets}
pairs={tuple(sorted(edge)) for edge in edges}
mutual=[p for p in pairs if p in edges and p[::-1] in edges]
mismatches={(d,o) for d,o in edges if (o,d) not in edges}
check('74 unordered pairs, 62 mutual, 12 document mismatches',len(pairs)==74 and len(mutual)==62 and len(mismatches)==12)
listed=load('journeyman-list-mismatches-'+SUFFIX)
check('Mismatch CSV matches directed calculation',{(r['destination_code'],r['origin_code']) for r in listed}==mismatches and len(listed)==12)
lengths=load('journeyman-list-lengths-'+SUFFIX)
check('17-row list-length subset matches source records',len(lengths)==17 and all(int(r['named_origins'])==len(jsets[r['destination_code']]) for r in lengths))
origins=load('journeyman-origin-counts-'+SUFFIX)
check('51-row origin subset including zeros',len(origins)==51 and all(int(r['named_on_selected_lists'])==frequency[r['origin_code']] for r in origins))
txj,txm=jsets['TX'],msets['TX']
check('Texas: 10 + 7 - 4 = 13',len(txj)==10 and len(txm)==7 and len(txj&txm)==4 and len(txj|txm)==13)
tx=load('texas-class-overlap-'+SUFFIX)
check('Texas overlap CSV matches both source lists',len(tx)==13 and all((r['on_journeyman_list']=='Yes')==(r['origin_code'] in txj) and (r['on_master_list']=='Yes')==(r['origin_code'] in txm) for r in tx))
check('Source and check date on every route', all(r['source_url'].startswith('https://') and r['checked_date']=='2026-09-29' for r in routes))
text=(ROOT/'electrical-license-reciprocity-by-state.md').read_text(encoding='utf-8')
body=text.split('---',2)[2]
blocks=re.findall(r'```DATASET\n([\s\S]*?)\n```',body)
check('Three preserved complete DATASET blocks',len(blocks)==3)
for name in (R,S,C):
    matching=[b for b in blocks if name in b]
    check('One dataset block for '+name,len(matching)==1)
    expected=load(name)
    header=','.join(expected[0])
    data=matching[0][matching[0].index(header):].strip()
    actual=list(csv.DictReader(io.StringIO(data)))
    check('Exact embedded CSV equality: '+name,actual==expected)
check('One BUILD SPEC, three CHART SPEC blocks',body.count('```BUILD SPEC\n')==1 and body.count('```CHART SPEC\n')==3)
visible=re.sub(r'```(?:DATASET|BUILD SPEC|CHART SPEC)\n[\s\S]*?\n```','',body)
ids=re.findall(r'<a id="([^"]+)"></a>',visible)
check('Unique explicit anchors',len(ids)==len(set(ids)))
links=re.findall(r'\]\(#([^\)]+)\)',visible)
check('All internal fragment links resolve',all(link in ids for link in links))
check('Single H1 and visible byline',len(re.findall(r'(?m)^# ',visible))==1 and len(re.findall(r'(?m)^By (?:the )?\[Castleport Test Prep Editorial Team\]',visible))==1)
check('Single research identity line',visible.count('Castleport Research is the research and reference section')==1)
for state in states:
    pattern=r'(?m)^### '+re.escape(state['state'])+r'\n\n([\s\S]*?)(?=\n\n\*Journeyman:)'
    match=re.search(pattern,visible)
    check('State profile matches CSV: '+state['code'],match is not None and match.group(1)==state['summary'])
faq=visible.split('## Frequently asked questions',1)[1].split('## Sources',1)[0]
check('Ten complete visible FAQs retained',len(re.findall(r'(?m)^### ',faq))==10)
for filename in re.findall(r'\]\(([^)]+\.(?:csv|png|py))\)',visible):
    if not filename.startswith(('http','/')):check('Local companion exists: '+filename,(ROOT/filename).is_file())
check('All chart images exist and are nonempty',all((ROOT/f'chart-{n}-{label}.png').stat().st_size>10000 for n,label in [(1,'class-gap'),(2,'list-length'),(3,'most-accepted')]))
result={
 'audit_date':'2026-09-29','scope':'Arithmetic and delivered-artifact consistency; not website or current-law testing',
 'counts':{'source_list_records':len(routes),'journey_entries':len(journey),'paired_journey_entries':paired_j,'master_destination_entries':len(master),'ratio_percent':round(100*len(master)/paired_j,1),'origin_jurisdictions_named':len(frequency),'zero_origins':51-len(frequency),'unordered_pairs':len(pairs),'mutual_pairs':len(mutual),'document_mismatches':len(mismatches),'texas_distinct_origins':len(txj|txm)},
 'checks_passed':len(checks),'checks':checks}
(ROOT/'validation-report.json').write_text(json.dumps(result,ensure_ascii=False,indent=2)+'\n',encoding='utf-8')
print(json.dumps({'checks_passed':len(checks),'counts':result['counts']},indent=2))
