#!/usr/bin/env python3
"""Ledger validator for the AI environmental-impact report.

    python3 validate.py ledger/
    python3 validate.py ledger/ --scan-prose draft/*.md

Exit 0 = no errors (warnings allowed). Exit 1 = at least one error.
Never publish while this exits 1. See conventions.publish_gate.

This is the only validator. A Node implementation existed until 2026-08-27 and was
removed: the site receives manual uploads and runs no build, so nothing needs Node.
"""
import json, os, re, sys
from datetime import date, datetime

errors, warnings = [], []
def E(where, msg): errors.append(f"{where}: {msg}")
def W(where, msg): warnings.append(f"{where}: {msg}")

# "9" and "roles" were added by dec-M; they sit after the main chapters and before the appendix.
CANONICAL_SECTIONS = ["0","1","2","3","4","5","6","7","8","9","roles","appendix","supplement"]
EXPLICIT_NULL_OK = {"superseded_by","previous_id","closed_at","resolution","editorial_label","source_id","resulting_decision"}
DATE_FIELDS = {"date","as_of","checked_at","verified_at","url_accessed","review_by","closed_at","generated_at"}
DATE_RE = re.compile(r"^\d{4}(-\d{2}(-\d{2})?)?$")  # day precision, or month where the source gave only a month
ID_PATTERNS = {
    "sources": r"^src-[a-z0-9]+(-[a-z0-9]+)*$",
    "measurements": r"^m-\d{3}$",
    "assumptions": r"^a-\d{3}$",
    "derivations": r"^der-\d{3}$",
    "disclosures": r"^disc-[a-z0-9_-]+(?:-r\d+)?$",
}


def load_ledger(target):
    if os.path.isfile(target):
        with open(target, encoding="utf-8") as f:
            return json.load(f), os.path.dirname(target) or "."
    files = sorted(f for f in os.listdir(target) if f.endswith(".json") and not f.startswith("."))
    if "index.json" not in files:
        E("layout", "split directory has no index.json")
    merged, index = {}, None
    for f in files:
        with open(os.path.join(target, f), encoding="utf-8") as fh:
            part = json.load(fh)
        if f == "index.json":
            index = part
            for k, v in part.items():
                if k != "files":
                    merged[k] = v
            continue
        for k, v in part.items():
            if k in merged:
                E("layout", f'key "{k}" defined in more than one file ({f})')
            merged[k] = v
    if index and index.get("files"):
        for declared in index["files"]:
            if not os.path.exists(os.path.join(target, declared)):
                E("layout", f"index.json declares missing file: {declared}")
        for f in files:
            if f != "index.json" and f not in index["files"]:
                W("layout", f"{f} is present but not declared in index.json")
    return merged, target


def walk(node, path, fn):
    fn(node, path)
    if isinstance(node, list):
        for i, v in enumerate(node):
            walk(v, f"{path}[{i}]", fn)
    elif isinstance(node, dict):
        for k, v in node.items():
            walk(v, f"{path}.{k}", fn)


def collect(ledger, key):
    v = ledger.get(key)
    return v if isinstance(v, list) else []


def id_set(ledger, key):
    return {r.get("id") for r in collect(ledger, key)}


def check_ids(ledger):
    for key, pattern in ID_PATTERNS.items():
        seen = set()
        for rec in collect(ledger, key):
            rid = rec.get("id")
            if not rid:
                E(key, "record without id"); continue
            if rid in seen:
                E(key, f"duplicate id {rid}")
            seen.add(rid)
            if rec.get("status") != "template" and not re.match(pattern, rid):
                W(key, f"id {rid} does not match the documented pattern {pattern}")
    notes = ledger.get("notes", {})
    note_patterns = {
        "decisions": r"^dec-[A-Z]$",
        "corrections": r"^c-\d{3}$",
        "rejected_sources": r"^rej-\d{3}$",
        "open_questions": r"^n-\d{3}$",
    }
    for sub in ("decisions","corrections","rejected_sources","open_questions"):
        seen = set()
        for rec in notes.get(sub, []) or []:
            rid = rec.get("id")
            if not rid:
                E(f"notes.{sub}", "record without id"); continue
            if rid in seen:
                E(f"notes.{sub}", f"duplicate id {rid}")
            seen.add(rid)
            if not re.match(note_patterns[sub], rid):
                E(f"notes.{sub}", f'id {rid} does not follow conventions.id_format ({note_patterns[sub]})')


def check_refs(ledger):
    sources = id_set(ledger, "sources")
    measurements = id_set(ledger, "measurements")
    assumptions = id_set(ledger, "assumptions")
    derivations = id_set(ledger, "derivations")
    notes = ledger.get("notes", {})
    decisions = {d.get("id") for d in notes.get("decisions", [])}

    def need(val, pool, where, what):
        if val is None:
            return
        if val not in pool:
            E(where, f'{what} points at "{val}", which does not exist')

    for m in collect(ledger, "measurements"):
        need(m.get("source_id"), sources, f"measurements.{m['id']}", "source_id")
        need(m.get("superseded_by"), measurements, f"measurements.{m['id']}", "superseded_by")
    for a in collect(ledger, "assumptions"):
        if a.get("source_id") is not None:
            need(a["source_id"], sources, f"assumptions.{a['id']}", "source_id")
        need(a.get("superseded_by"), assumptions, f"assumptions.{a['id']}", "superseded_by")
    for d in collect(ledger, "derivations"):
        for inp in d.get("inputs", []):
            ref = inp.get("ref")
            if ref not in measurements and ref not in assumptions and ref not in derivations:
                E(f"derivations.{d['id']}", f'input ref "{ref}" does not exist')
        need(d.get("boundary_inherited_from"), measurements, f"derivations.{d['id']}", "boundary_inherited_from")
        need(d.get("misuse_guard_inherited_from"), measurements, f"derivations.{d['id']}", "misuse_guard_inherited_from")
        need(d.get("superseded_by"), derivations, f"derivations.{d['id']}", "superseded_by")
    disclosures = id_set(ledger, "disclosures")
    for disc in collect(ledger, "disclosures"):
        for s in disc.get("source_ids", []):
            need(s, sources, f"disclosures.{disc['id']}", "source_ids")
        need(disc.get("superseded_by"), disclosures, f"disclosures.{disc['id']}", "superseded_by")
    for c in notes.get("corrections", []):
        need(c.get("resulting_decision"), decisions, f"notes.corrections.{c['id']}", "resulting_decision")
    for q in notes.get("open_questions", []):
        need(q.get("resolution"), decisions, f"notes.open_questions.{q['id']}", "resolution")
        has_resolution = bool(q.get("resolution"))
        has_finding = bool(q.get("resolution_finding"))
        if q.get("state") == "closed" and not has_resolution and not has_finding:
            E(f"notes.open_questions.{q['id']}", "closed with neither a resolution (decision id) nor a resolution_finding")
        if q.get("state") == "closed" and has_resolution and has_finding:
            E(f"notes.open_questions.{q['id']}", "closed with both resolution and resolution_finding; choose the closure type")
        if q.get("state") == "open" and (has_resolution or has_finding):
            E(f"notes.open_questions.{q['id']}", "open but already has a resolution or resolution_finding")
    for d in notes.get("decisions", []):
        need(d.get("superseded_by"), decisions, f"notes.decisions.{d['id']}", "superseded_by")

    sections = set((ledger.get("state") or {}).get("sections", {}).keys())
    for d in notes.get("decisions", []):
        for s in d.get("affects", []):
            if s not in sections:
                E(f"notes.decisions.{d['id']}", f'affects unknown section "{s}"')
    for q in notes.get("open_questions", []):
        for sid in q.get("source_ids", []):
            need(sid, sources, f"notes.open_questions.{q['id']}", "source_ids")
        for s in q.get("blocking", []):
            if s not in sections:
                E(f"notes.open_questions.{q['id']}", f'blocking unknown section "{s}"')


def check_nulls_have_state(ledger):
    def fn(node, path):
        if not isinstance(node, dict):
            return
        for k, v in node.items():
            if v is not None or k in EXPLICIT_NULL_OK:
                continue
            if "state" in node or "status" in node or f"{k}_state" in node:
                continue
            E(f"{path}.{k}", "null without an adjacent state field (see conventions.null_policy)")
    walk(ledger, "$", fn)


def check_knowledge_states(ledger):
    allowed = set((ledger.get("taxonomy") or {}).get("knowledge_state_values", {}).keys())
    if not allowed:
        E("taxonomy", "knowledge_state_values missing"); return
    question_states = {"open","closed"}
    question_record = re.compile(r"open_questions\[\d+\]$")

    def fn(node, path):
        if not isinstance(node, dict) or not isinstance(node.get("state"), str):
            return
        # only the question record itself uses open/closed; nested objects use knowledge states
        if question_record.search(path):
            if node["state"] not in question_states:
                E(f"{path}.state", f'open_questions state must be open|closed, got "{node["state"]}"')
            return
        if node["state"] not in allowed and node["state"] != "not_applicable":
            E(f"{path}.state", f'unknown knowledge state "{node["state"]}"')
    walk(ledger, "$", fn)


def check_date_formats(ledger):
    def fn(node, path):
        # taxonomy and conventions hold definitions of these field names, not values of them
        if not isinstance(node, dict) or path.startswith("$.taxonomy") or path.startswith("$.conventions"):
            return
        for k, v in node.items():
            if k not in DATE_FIELDS:
                continue
            val = v.get("value") if isinstance(v, dict) and "value" in v else v
            if val is None:
                continue
            if not isinstance(val, str) or not DATE_RE.match(val):
                E(f"{path}.{k}", f'date must be YYYY-MM-DD (conventions.date_format), got "{val}"')
                continue
            try:
                padded = val + "-01" * ((10 - len(val)) // 3)
                date.fromisoformat(padded)
            except ValueError:
                E(f"{path}.{k}", f'"{val}" is not a real calendar date')
    walk(ledger, "$", fn)


def check_breakdowns(ledger):
    comps = set((ledger.get("taxonomy") or {}).get("components", {}).keys())
    for m in collect(ledger, "measurements"):
        q = m.get("quantity", {})
        breakdown = q.get("breakdown")
        if not isinstance(breakdown, list):
            continue
        target_path = q.get("breakdown_target")
        if not target_path:
            E(f"measurements.{m['id']}", "breakdown present but breakdown_target missing"); continue
        target = m  # breakdown_target is a dotted path from the measurement root
        for seg in target_path.split("."):
            target = (target or {}).get(seg) if isinstance(target, dict) else None
        if not isinstance(target, dict) or not isinstance(target.get("value"), (int, float)):
            E(f"measurements.{m['id']}", f'breakdown_target "{target_path}" does not resolve to a number'); continue
        try:
            total = sum(b["value"] for b in breakdown)
        except (KeyError, TypeError):
            E(f"measurements.{m['id']}", "breakdown contains a non-numeric value"); continue
        delta = abs(total - target["value"])
        if delta > 0.05:
            E(f"measurements.{m['id']}", f'breakdown sums to {total} but {target_path} is {target["value"]} (delta {delta:.3f})')
        for b in breakdown:
            if b.get("component") not in comps:
                E(f"measurements.{m['id']}", f'breakdown component "{b.get("component")}" is not in taxonomy.components')
            if b.get("provenance") not in ("measured","estimated","unknown"):
                E(f"measurements.{m['id']}", f'breakdown component "{b.get("component")}" has provenance "{b.get("provenance")}"')


def check_boundaries(ledger):
    tax = ledger.get("taxonomy") or {}
    tax_comps = list(tax.get("components", {}).keys())
    statuses = set(tax.get("component_status_values", {}).keys())
    tiers = set(tax.get("boundary_tiers", {}).keys())
    if not tax_comps:
        E("taxonomy", "components missing"); return
    for m in collect(ledger, "measurements"):
        b = m.get("boundary")
        if not b:
            E(f"measurements.{m['id']}", "boundary missing"); continue
        if b.get("tier") not in tiers:
            E(f"measurements.{m['id']}", f'boundary.tier "{b.get("tier")}" is not a taxonomy tier')
        got = list((b.get("components") or {}).keys())
        for c in tax_comps:
            if c not in got:
                E(f"measurements.{m['id']}", f'boundary.components is missing "{c}" — an omitted component must not be read as unknown')
        for c in got:
            if c not in tax_comps:
                E(f"measurements.{m['id']}", f'boundary.components has unknown component "{c}"')
            elif b["components"][c] not in statuses:
                E(f"measurements.{m['id']}", f'component "{c}" has status "{b["components"][c]}"')
        for c in (b.get("component_notes") or {}):
            if c not in tax_comps:
                E(f"measurements.{m['id']}", f'component_notes references unknown component "{c}"')


def check_misuse_guards(ledger):
    for m in collect(ledger, "measurements"):
        g = m.get("misuse_guard")
        if not g:
            E(f"measurements.{m['id']}", "misuse_guard is required and missing"); continue
        if not g.get("not_applicable_to"):
            E(f"measurements.{m['id']}", "misuse_guard.not_applicable_to is empty")
        if not g.get("forbidden_claims"):
            E(f"measurements.{m['id']}", "misuse_guard.forbidden_claims is empty")
        for fc in g.get("forbidden_claims", []):
            if not fc.get("en"):
                E(f"measurements.{m['id']}", "a forbidden_claim has no text")
    for d in collect(ledger, "derivations"):
        if not d.get("misuse_guard") and not d.get("misuse_guard_inherited_from"):
            E(f"derivations.{d['id']}", "needs either misuse_guard or misuse_guard_inherited_from")


def check_numeric_use(ledger):
    """A display-only measurement must never feed a calculation."""
    forbidden = {m["id"] for m in collect(ledger, "measurements") if m.get("numeric_use") == "forbidden"}
    allowed_values = set((ledger.get("taxonomy") or {}).get("numeric_use_values", {}).keys())
    for m in collect(ledger, "measurements"):
        nu = m.get("numeric_use")
        if not nu:
            E(f"measurements.{m['id']}", "numeric_use missing (allowed | forbidden)")
        elif allowed_values and nu not in allowed_values:
            E(f"measurements.{m['id']}", f'numeric_use "{nu}" is not in taxonomy.numeric_use_values')
    for d in collect(ledger, "derivations"):
        for inp in d.get("inputs", []):
            if inp.get("ref") in forbidden:
                E(f"derivations.{d['id']}", f'input ref "{inp["ref"]}" is numeric_use=forbidden and must not feed a calculation')
        if d.get("boundary_inherited_from") in forbidden:
            E(f"derivations.{d['id']}", f'boundary_inherited_from "{d["boundary_inherited_from"]}" is numeric_use=forbidden')


def check_measurement_status(ledger):
    allowed = set((ledger.get("taxonomy") or {}).get("measurement_status_values", {}).keys())
    for m in collect(ledger, "measurements"):
        if allowed and m.get("status") not in allowed:
            E(f"measurements.{m['id']}", f'status "{m.get("status")}" is not in taxonomy.measurement_status_values')
        if m.get("publishable") is None:
            E(f"measurements.{m['id']}", "publishable missing (true | false)")
        if m.get("status") == "draft_pending_verification" and m.get("publishable"):
            E(f"measurements.{m['id']}", "draft_pending_verification cannot be publishable")


def check_units(ledger):
    units = set((ledger.get("taxonomy") or {}).get("units", {}).keys())
    if not units:
        E("taxonomy", "units missing"); return
    for m in collect(ledger, "measurements"):
        u = (m.get("quantity") or {}).get("unit")
        if u not in units:
            E(f"measurements.{m['id']}", f'quantity.unit "{u}" is not in taxonomy.units')
    for d in collect(ledger, "derivations"):
        u = (d.get("result") or {}).get("unit")
        if u not in units:
            E(f"derivations.{d['id']}", f'result.unit "{u}" is not in taxonomy.units')
        src_units = {(next((m for m in collect(ledger, "measurements") if m["id"] == i.get("ref")), {})
                      .get("quantity") or {}).get("unit")
                     for i in d.get("inputs", [])}
        src_units.discard(None)
        if src_units and u not in src_units and not d.get("unit_conversion"):
            E(f"derivations.{d['id']}", f'result.unit "{u}" differs from its inputs {sorted(src_units)} with no unit_conversion recorded')


def check_verification(ledger):
    people = (ledger.get("conventions") or {}).get("participants", {})
    depths = set((ledger.get("taxonomy") or {}).get("verification_depth_values", {}).keys())
    for key in ("measurements","derivations"):
        for r in collect(ledger, key):
            for f in ("verified_by","verified_at","verification_depth"):
                if not r.get(f):
                    E(f"{key}.{r['id']}", f"{f} missing")
            vb = r.get("verified_by")
            if vb and vb not in people:
                E(f"{key}.{r['id']}", f'verified_by "{vb}" is not a declared participant')
            vd = r.get("verification_depth")
            if vd and vd not in depths:
                E(f"{key}.{r['id']}", f'verification_depth "{vd}" is not in taxonomy')


def check_disclosures(ledger, today):
    tax = ledger.get("taxonomy") or {}
    limit = (tax.get("staleness_policy") or {}).get("disclosures_days", 90)
    items = set(tax.get("disclosure_items", {}).keys())
    for d in collect(ledger, "disclosures"):
        if d.get("item") not in items:
            E(f"disclosures.{d['id']}", f'item "{d.get("item")}" is not in taxonomy.disclosure_items')
        allowed = set(tax.get("disclosure_status_values", {}).keys())
        companies = (tax.get("surveyed_companies") or {}).get("companies", [])
        if companies and d.get("company") not in companies:
            E(f"disclosures.{d['id']}", f'company "{d.get("company")}" is not in taxonomy.surveyed_companies')
        expected_id = f"disc-{d.get('company')}-{d.get('item')}"
        if not (d.get("id") == expected_id or re.match(rf"^{re.escape(expected_id)}-r\d+$", d.get("id") or "")):
            E(f"disclosures.{d['id']}", f'id does not match company/item (expected {expected_id} or revision suffix -rN)')
        req = ((tax.get("disclosure_items") or {}).get(d.get("item")) or {}).get("definition_required", [])
        missing = [k for k in req if k not in (d.get("definition_completeness") or {})]
        if missing:
            E(f"disclosures.{d['id']}", f'definition_completeness is missing {", ".join(missing)}')
        if allowed and d.get("disclosure_status") not in allowed:
            E(f"disclosures.{d['id']}", f'disclosure_status "{d.get("disclosure_status")}" is not one of the declared values')
        as_of = d.get("checked_at")
        as_of_value = as_of.get("value") if isinstance(as_of, dict) else as_of
        if as_of_value is None:
            E(f"disclosures.{d['id']}", "checked_at is required on every cell, including uninvestigated ones")
        else:
            try:
                as_of_date = date.fromisoformat(as_of_value + "-01" * ((10 - len(as_of_value)) // 3))
            except (ValueError, TypeError):
                E(f"disclosures.{d['id']}", f'checked_at "{as_of_value}" is not a usable date'); continue
            age = (today - as_of_date).days
            if age > limit:
                W(f"disclosures.{d['id']}", f"last checked {age} days ago (limit {limit}) — re-verify")
        rec_states = set(tax.get("disclosure_record_status_values", {}).keys())
        if rec_states and d.get("status") not in rec_states:
            E(f"disclosures.{d['id']}", f'status "{d.get("status")}" is not in taxonomy.disclosure_record_status_values')
        if d.get("publishable") is None:
            E(f"disclosures.{d['id']}", "publishable missing (true | false)")
        negative = d.get("knowledge_state") == "investigated_not_found"
        if negative and not d.get("search_paths_taken"):
            E(f"disclosures.{d['id']}", "investigated_not_found requires search_paths_taken (see taxonomy.negative_finding_rule)")
        if d.get("disclosure_status") != "not_investigated" and not d.get("source_ids") and not negative:
            if d.get("status") != "draft_pending_source" or d.get("source_state") != "not_extracted" or d.get("publishable") is not False:
                E(f"disclosures.{d['id']}",
                  "has no source_ids; a sourceless cell must be status=draft_pending_source with source_state=not_extracted and publishable=false")
        if d.get("publishable") and not d.get("source_ids") and not negative:
            E(f"disclosures.{d['id']}", "publishable=true with no source_ids")
        people = (ledger.get("conventions") or {}).get("participants", {})
        if d.get("checked_by") and d["checked_by"] not in people:
            E(f"disclosures.{d['id']}", f'checked_by "{d["checked_by"]}" is not a declared participant')
        for k, v in (d.get("definition_completeness") or {}).items():
            if not isinstance(v, dict) or "answer" not in v or "state" not in v:
                E(f"disclosures.{d['id']}", f'definition_completeness.{k} must be an object with answer and state')
        vt = (tax.get("venue_types") or {})
        for ii in d.get("integrity_issues", []):
            if not ii.get("id") or not ii.get("severity"):
                E(f"disclosures.{d['id']}", "an integrity_issue is missing id or severity")


def check_decisions(ledger):
    notes = ledger.get("notes", {})
    people = (ledger.get("conventions") or {}).get("participants", {})
    root_cause_values = set((ledger.get("taxonomy") or {}).get("correction_root_cause_values", {}).keys())
    for d in notes.get("decisions", []):
        rc = d.get("reversal_condition")
        if not rc or not rc.get("en"):
            E(f"notes.decisions.{d['id']}", 'reversal_condition is required (write "None" explicitly if there is none)')
        if not d.get("raised_by"):
            E(f"notes.decisions.{d['id']}", "raised_by missing")
        for f in ("raised_by","accepted_by"):
            if d.get(f) and d[f] not in people:
                E(f"notes.decisions.{d['id']}", f'{f} "{d[f]}" is not a declared participant')
    for c in notes.get("corrections", []):
        if c.get("raised_by") and c["raised_by"] not in people:
            E(f"notes.corrections.{c['id']}", f'raised_by "{c["raised_by"]}" is not a declared participant')
        if c.get("root_cause") and c["root_cause"] not in people and c["root_cause"] not in root_cause_values:
            E(f"notes.corrections.{c['id']}", f'root_cause "{c["root_cause"]}" is neither a declared participant nor a taxonomy.correction_root_cause_values entry')
        if c.get("made_by") and c["made_by"] not in people:
            E(f"notes.corrections.{c['id']}", f'made_by "{c["made_by"]}" is not a declared participant')
        for p in c.get("corrected_by", []):
            if p not in people:
                E(f"notes.corrections.{c['id']}", f'corrected_by "{p}" is not a declared participant')


def check_editorial_labels(ledger):
    """An editorial label must never be silently readable as a different record's ledger id."""
    decs = ledger.get("notes", {}).get("decisions", [])
    ids = {d.get("id") for d in decs}
    for d in decs:
        lab, rid = d.get("editorial_label"), d.get("id")
        if not lab:
            continue
        if lab in ids and lab != rid:
            E(f"notes.decisions.{rid}", f'editorial_label "{lab}" is also another record\'s ledger id')
        if "-" not in lab:
            continue
        implied = "dec-" + lab.split("-", 1)[1]
        if implied not in ids or implied == rid:
            continue
        msg = f'editorial_label "{lab}" reads as ledger id "{implied}", which is a different decision'
        if d.get("editorial_label_warning"):
            W(f"notes.decisions.{rid}", msg + " — cite ledger ids, not editorial labels")
        else:
            E(f"notes.decisions.{rid}", msg + " — editorial_label_warning is required")


def check_forbidden_words(ledger):
    fw = (ledger.get("editorial_rules") or {}).get("forbidden_words")
    if not fw:
        E("editorial_rules", "forbidden_words missing"); return
    if "en" not in fw:
        E("editorial_rules", "forbidden_words.en is missing")
    for lang in [k for k in ("ja", "en") if k in fw]:
        if not fw.get(lang):
            E("editorial_rules", f"forbidden_words.{lang} is empty"); continue
        for t in fw[lang]:
            if not (t.get("term") and t.get("regex") and t.get("reason")):
                E("editorial_rules", f"forbidden_words.{lang} entry is missing term/regex/reason")
            try:
                re.compile(t.get("regex", ""))
            except re.error:
                E("editorial_rules", f'forbidden_words.{lang} regex does not compile: {t.get("regex")}')


def check_state(ledger):
    s = ledger.get("state")
    if not s:
        E("state", "missing"); return
    keys = list((s.get("sections") or {}).keys())
    missing = [k for k in CANONICAL_SECTIONS if k not in keys]
    extra = [k for k in keys if k not in CANONICAL_SECTIONS]
    if missing:
        E("state.sections", "missing sections: " + ", ".join(missing))
    if extra:
        E("state.sections", "unexpected sections: " + ", ".join(extra))
    if len(s.get("next_actions", [])) > 3:
        E("state.next_actions", f"{len(s['next_actions'])} entries; the convention is at most 3")
    if not s.get("read_order_for_new_session"):
        E("state", "read_order_for_new_session is required")


def check_sources(ledger):
    rb = (ledger.get("taxonomy") or {}).get("reported_by_values", [])
    for s in collect(ledger, "sources"):
        labels = [v.get("label") for v in s.get("versions", [])]
        rv = s.get("read_version")
        if isinstance(rv, dict):
            rv = None
        if rv and rv not in labels:
            E(f"sources.{s['id']}", f'read_version "{rv}" is not among versions [{", ".join(str(l) for l in labels)}]')
        if not s.get("url_accessed"):
            E(f"sources.{s['id']}", "url_accessed missing")
        dv_allowed = set((ledger.get("taxonomy") or {}).get("date_verification_values", {}).keys())
        if not s.get("date_verification"):
            E(f"sources.{s['id']}", "date_verification missing — where the publication date came from must be stated")
        elif dv_allowed and s["date_verification"] not in dv_allowed:
            E(f"sources.{s['id']}", f'date_verification "{s["date_verification"]}" is not in taxonomy')
        all_ii = ({i["id"] for src in collect(ledger, "sources") for i in src.get("integrity_issues", [])}
                  | {i["id"] for dd in collect(ledger, "disclosures") for i in dd.get("integrity_issues", [])})
        for ref in s.get("supports", []):
            if ref not in all_ii:
                E(f"sources.{s['id']}", f'supports "{ref}", which is not an existing integrity_issue id')
        vts = set((ledger.get("taxonomy") or {}).get("venue_types", {}).keys())
        vt = (s.get("publication") or {}).get("venue_type")
        if vts and vt not in vts:
            E(f"sources.{s['id']}", f'venue_type "{vt}" is not in taxonomy.venue_types')
        dts = set((ledger.get("taxonomy") or {}).get("document_types", {}).keys())
        if dts and s.get("document_type") not in dts:
            E(f"sources.{s['id']}", f'document_type "{s.get("document_type")}" is not in taxonomy.document_types')
        if rb and s.get("reported_by") not in rb:
            E(f"sources.{s['id']}", f'reported_by "{s.get("reported_by")}" is not in taxonomy')


def check_leaks(ledger, root):
    clone = json.loads(json.dumps(ledger))
    guard_cfg = (clone.get("conventions") or {}).get("leak_guard") or {}
    if "leak_guard" in (clone.get("conventions") or {}):
        del clone["conventions"]["leak_guard"]
    text = json.dumps(clone, ensure_ascii=False)

    for p in [r"/Users/", r"/home/", r"[A-Za-z]:\\\\", r"file://", r"/private/tmp/", r"/var/folders/", r"~/[A-Za-z]"]:
        hit = re.search(p, text)
        if hit:
            E("leak_guard", f"local filesystem path found in ledger content: {hit.group(0)}")

    guard_file = guard_cfg.get("nickname_blocklist_file")
    if not guard_file:
        W("leak_guard", "no nickname_blocklist_file declared"); return
    guard_path = os.path.join(root, guard_file)
    if not os.path.exists(guard_path):
        E("leak_guard", f"{guard_file} not found — the nickname scan cannot run. Create it (an empty nicknames list is a valid, explicit answer).")
        return
    with open(guard_path, encoding="utf-8") as f:
        guard = json.load(f)
    for nick in guard.get("nicknames", []):
        if not nick:
            continue
        # ASCII entries match on word boundaries, so a short entry cannot fire inside a longer word.
        # Non-ASCII nicknames have no word boundaries, so they match as substrings.
        if nick.isascii():
            hit = re.search(r"\b" + re.escape(nick) + r"\b", text, re.IGNORECASE)
        else:
            hit = re.search(re.escape(nick), text)
        if hit:
            E("leak_guard", "blocked nickname appears in ledger content")


def scan_prose(ledger, files):
    fw = (ledger.get("editorial_rules") or {}).get("forbidden_words")
    if not fw:
        return
    hits = 0
    for path in files:
        if not os.path.exists(path):
            W("scan-prose", f"no such file: {os.path.basename(path)}"); continue
        lang = "en" if ".en." in path else "ja"
        with open(path, encoding="utf-8") as f:
            lines = f.read().split("\n")
        for t in fw.get(lang, []):
            rx = re.compile(t["regex"])
            for i, line in enumerate(lines, 1):
                if rx.search(line):
                    hits += 1
                    label = f"{os.path.basename(path)}:{i}"
                    if t.get("high_false_positive"):
                        W("prose", f'{label} "{t["term"]}" ({t["reason"]}) — high false-positive term, human review')
                    else:
                        E("prose", f'{label} "{t["term"]}" ({t["reason"]})')
    print(f"  scanned {len(files)} prose file(s), {hits} hit(s)")


def main():
    args = sys.argv[1:]
    if not args:
        print("usage: python3 validate.py <ledger.json | ledger/> [--scan-prose <files...>]", file=sys.stderr)
        sys.exit(2)
    target = args[0]
    prose = args[args.index("--scan-prose") + 1:] if "--scan-prose" in args else []

    ledger, root = load_ledger(target)
    today = date.fromisoformat(ledger["generated_at"]) if ledger.get("generated_at") else date.today()

    check_ids(ledger)
    check_refs(ledger)
    check_nulls_have_state(ledger)
    check_knowledge_states(ledger)
    check_date_formats(ledger)
    check_breakdowns(ledger)
    check_boundaries(ledger)
    check_misuse_guards(ledger)
    check_numeric_use(ledger)
    check_measurement_status(ledger)
    check_units(ledger)
    check_verification(ledger)
    check_disclosures(ledger, today)
    check_decisions(ledger)
    check_editorial_labels(ledger)
    check_forbidden_words(ledger)
    check_state(ledger)
    check_sources(ledger)
    check_leaks(ledger, root)
    if prose:
        scan_prose(ledger, prose)

    pending = [d["id"] for d in collect(ledger, "disclosures") if d.get("status") == "draft_pending_source"]
    if pending:
        W("draft_pending_source", f"{len(pending)} cell(s) held back for a missing source record: " + ", ".join(pending))

    print(f"ledger: {target}")
    print(f'  sources {len(collect(ledger,"sources"))} / measurements {len(collect(ledger,"measurements"))} / '
          f'assumptions {len(collect(ledger,"assumptions"))} / derivations {len(collect(ledger,"derivations"))} / '
          f'disclosures {len(collect(ledger,"disclosures"))} '
          f'(draft_pending_source {len([d for d in collect(ledger,"disclosures") if d.get("status") == "draft_pending_source"])})')
    for w in warnings:
        print(f"  WARN  {w}")
    for e in errors:
        print(f"  ERROR {e}")
    print(f"OK ({len(warnings)} warning(s))" if not errors
          else f"FAILED ({len(errors)} error(s), {len(warnings)} warning(s))")
    sys.exit(0 if not errors else 1)


if __name__ == "__main__":
    main()
