#!/usr/bin/env python3
"""Recalculate and audit the public aggregate results."""

from __future__ import annotations

import csv
import fnmatch
import html
import json
import re
import sys
from collections import Counter, defaultdict
from pathlib import Path


ROOT = Path(__file__).resolve().parents[1]
DATA = ROOT / "02-data"
DESIGN = ROOT / "01-study-design"
RESULTS = ROOT / "04-results"
TABLES = RESULTS / "generated-tables"
FIGURES = RESULTS / "generated-figures"
GOVERNANCE = ROOT / "06-governance"
INSTRUMENTS = ROOT / "05-instruments"

EXPECTED_RQS = {
    "RQ1": "How were documented needs, constraints, and opportunities from one edition translated into program changes in the next, while preserving the school's founding educational objectives?",
    "RQ2a": "How were disciplinary perspectives, professional expertise, practical activities, responsible innovation, and networking jointly configured across the two editions?",
    "RQ2b": "What participant reactions were reported in relation to this program configuration?",
    "RQ3": "How was international participation supported from outreach to on site delivery, and which organizational resources and design choices enabled this process?",
}


def read_csv(path: Path) -> list[dict[str, str]]:
    with path.open(newline="", encoding="utf-8") as handle:
        return list(csv.DictReader(handle))


def integer(row: dict[str, str], key: str) -> int:
    return int(row[key])


def pct(value: float, decimals: int = 1) -> str:
    result = f"{100.0 * value:.{decimals}f}"
    if result.endswith(".0") and value in (0.0, 1.0):
        return result[:-2]
    return result


def tex_escape(value: str) -> str:
    replacements = {
        "\\": r"\textbackslash{}",
        "&": r"\&",
        "%": r"\%",
        "$": r"\$",
        "#": r"\#",
        "_": r"\_",
        "{": r"\{",
        "}": r"\}",
        "~": r"\textasciitilde{}",
        "^": r"\textasciicircum{}",
    }
    return "".join(replacements.get(char, char) for char in value)


def paper_cell(value: str) -> str:
    """Format a prose CSV field exactly as it appears in a manuscript table cell."""
    return tex_escape(value.strip().removesuffix("."))


def paper_code_list(value: str) -> str:
    codes = [code.strip() for code in value.split(",")]
    if len(codes) == 1:
        return codes[0]
    if len(codes) == 2:
        return f"{codes[0]} and {codes[1]}"
    return f"{', '.join(codes[:-1])}, and {codes[-1]}"


def transition_content(node_id: str, value: str) -> str:
    escaped = tex_escape(value)
    if node_id in {"initial", "redesign", "second"}:
        for term, code in (
            ("Geospatial Data and Quantitative Analysis in Social Science", "GEO"),
            ("Artificial Intelligence", "AI"),
            ("Digital Twins", "DT"),
            ("Robotics", "ROB"),
        ):
            escaped = re.sub(rf"\b{re.escape(term)}\b", lambda _: rf"\emph{{{code}}}", escaped)
    return escaped


def paper_track_terms(value: str) -> str:
    formatted = paper_cell(value)
    for term, code in (
        ("Geospatial Data and Quantitative Analysis in Social Science", "GEO"),
        ("Artificial Intelligence", "AI"),
        ("Digital Twins", "DT"),
        ("Robotics", "ROB"),
        ("GEO", "GEO"),
        ("AI", "AI"),
    ):
        formatted = re.sub(rf"\b{re.escape(term)}\b", lambda _: rf"\emph{{{code}}}", formatted)
    return formatted


class Audit:
    def __init__(self) -> None:
        self.items: list[tuple[str, bool, str]] = []

    def check(self, name: str, condition: bool, detail: str) -> None:
        self.items.append((name, bool(condition), detail))

    @property
    def passed(self) -> bool:
        return all(ok for _, ok, _ in self.items)


def administrative_index(rows: list[dict[str, str]]) -> dict[str, dict[str, str]]:
    return {row["row_id"]: row for row in rows}


def dictionary_field_matches(field_spec: str, headers: set[str]) -> bool:
    if ".." in field_spec:
        start, end = field_spec.split("..", 1)
        start_match = re.fullmatch(r"(.*?)(\d+)", start)
        end_match = re.fullmatch(r"(.*?)(\d+)", end)
        if not start_match or not end_match or start_match.group(1) != end_match.group(1):
            return False
        prefix = start_match.group(1)
        expected = {f"{prefix}{index}" for index in range(int(start_match.group(2)), int(end_match.group(2)) + 1)}
        return expected.issubset(headers)
    if "*" in field_spec:
        return any(fnmatch.fnmatch(header, field_spec) for header in headers)
    return field_spec in headers


def validate_data(
    admin: list[dict[str, str]],
    tracks: list[dict[str, str]],
    track_combinations: list[dict[str, str]],
    preference_admission: list[dict[str, str]],
    scale: list[dict[str, str]],
    track_attendance: list[dict[str, str]],
    binary: list[dict[str, str]],
    likert: list[dict[str, str]],
    categorical: list[dict[str, str]],
    first_workbook_check: list[dict[str, str]],
    attendance_patterns: list[dict[str, str]],
    country_summary: list[dict[str, str]],
    profiles: list[dict[str, str]],
    demographics: list[dict[str, str]],
    agenda: list[dict[str, str]],
    slots: list[dict[str, str]],
    agenda_coding: list[dict[str, str]],
    learning_content: list[dict[str, str]],
    speaker_composition: list[dict[str, str]],
    objectives: list[dict[str, str]],
    decisions: list[dict[str, str]],
    protocol: list[dict[str, str]],
    comparators: list[dict[str, str]],
    traceability: list[dict[str, str]],
    rq1_themes: list[dict[str, str]],
    evidence: list[dict[str, str]],
    data_dictionary: list[dict[str, str]],
    questionnaire_2025: list[dict[str, str]],
    questionnaire_2026: list[dict[str, str]],
    questionnaire_common: list[dict[str, str]],
    open_comment_inventory: list[dict[str, str]],
    access_process: list[dict[str, str]],
    capacity_pressure: list[dict[str, str]],
    scholarship_model: list[dict[str, str]],
    scholarship_rule: list[dict[str, str]],
    speaker_by_track: list[dict[str, str]],
    propositions: list[dict[str, str]],
    output_traceability: list[dict[str, str]],
    explanatory_synthesis: list[dict[str, str]],
    curricular_connections: list[dict[str, str]],
    corpus_metadata: list[dict[str, str]],
    speaker_detailed: list[dict[str, str]],
    networking_design: list[dict[str, str]],
    communication_use: list[dict[str, str]],
    rover_design: list[dict[str, str]],
    expertise_fields: list[dict[str, str]],
    source_schemas: list[dict[str, str]],
    transition_nodes: list[dict[str, str]],
    transition_edges: list[dict[str, str]],
    paper_objectives: list[dict[str, str]],
    paper_corpus: list[dict[str, str]],
    paper_participation: list[dict[str, str]],
    rq_document: str,
) -> Audit:
    audit = Audit()
    a = administrative_index(admin)

    required_admin_ids = {
        "A001", "A002", "A003", "A004", "A005", "A006", "A008", "A102", "A103", "A104", "A106",
        "A107", "A108", "A109", "A110", "A111", "A112", "A113", "A114", "A115", "A116",
        "A117", "A118", "A119", "A120", "A121", "A122", "A123", "A124", "A125", "A126",
        "A128", "A130", "A132", "A133", "A134", "A135",
    }
    audit.check("administrative row IDs unique", len(a) == len(admin), f"rows={len(admin)}; unique IDs={len(a)}")
    audit.check("administrative rows required by generated outputs", required_admin_ids.issubset(a), f"missing={sorted(required_admin_ids - set(a))}")
    audit.check("administrative numeric fields parse", all(row["value_numeric"].isdigit() for row in admin), "all released values are non-negative integers")
    audit.check("first registration covers attendance", integer(a["A002"], "value_numeric") >= integer(a["A003"], "value_numeric"), f"registered={a['A002']['value_text']}; attendance={a['A003']['value_text']}")
    audit.check("first questionnaire rows bounded by attendance", integer(a["A006"], "value_numeric") <= integer(a["A003"], "value_numeric"), f"questionnaire rows={a['A006']['value_text']}; attendance={a['A003']['value_text']}")
    audit.check("later source-local de-duplication relations", integer(a["A102"], "value_numeric") >= integer(a["A103"], "value_numeric") and integer(a["A104"], "value_numeric") >= integer(a["A106"], "value_numeric") and integer(a["A119"], "value_numeric") >= integer(a["A120"], "value_numeric") and integer(a["A121"], "value_numeric") >= integer(a["A122"], "value_numeric"), "rows/submissions are not fewer than source-local normalized keys")
    audit.check("later selection nesting within decision IDs", integer(a["A104"], "value_numeric") >= integer(a["A108"], "value_numeric") >= integer(a["A110"], "value_numeric"), f"decision={a['A104']['value_text']}; admitted={a['A108']['value_text']}; scholarship-selected={a['A110']['value_text']}")
    audit.check("weekly recollections do not exceed working cap", all(integer(a[row_id], "value_numeric") <= integer(a["A126"], "value_numeric") for row_id in ("A123", "A124", "A125")), f"headcounts={a['A123']['value_text']}/{a['A124']['value_text']}/{a['A125']['value_text']}; cap={a['A126']['value_text']}")
    audit.check("local enrollments preserve person and track units", integer(a["A134"], "value_numeric") == 7 and integer(a["A135"], "value_numeric") == 16 and integer(a["A135"], "value_numeric") >= integer(a["A134"], "value_numeric"), f"local people={a['A134']['value_text']}; local track enrollments={a['A135']['value_text']}")
    audit.check("second scholarship field is explicitly selection-stage", "selection-stage" in a["A133"]["measure"] and "not proof" in a["A133"]["interpretation_boundary"], a["A133"]["interpretation_boundary"])

    scale_by_id = {row["indicator_id"]: row for row in scale}
    expected_scale_ids = {f"SC{index:02d}" for index in range(1, 7)}
    audit.check("program scale indicator IDs", set(scale_by_id) == expected_scale_ids, f"observed={sorted(scale_by_id)}")
    for row in scale:
        first_value = float(row["first_value"])
        later_value = float(row["later_value"])
        expected_index = round(100 * later_value / first_value, 1)
        expected_change = round(100 * (later_value - first_value) / first_value, 1)
        audit.check(
            f"program scale arithmetic {row['indicator_id']}",
            float(row["later_index_first_100"]) == expected_index and float(row["percent_change"]) == expected_change,
            f"index={expected_index}; change={expected_change}",
        )

    attendance_by_edition: dict[str, list[dict[str, str]]] = defaultdict(list)
    for row in track_attendance:
        attendance_by_edition[row["edition"]].append(row)
    audit.check("track attendance editions", set(attendance_by_edition) == {"2025", "2026"}, f"editions={sorted(attendance_by_edition)}")
    audit.check("track attendance rows per edition", all(len(rows) == 3 for rows in attendance_by_edition.values()), f"rows={[(edition, len(rows)) for edition, rows in sorted(attendance_by_edition.items())]}")
    first_track_counts = {
        row["track"]: integer(row, "count")
        for row in tracks
        if row["edition"] == "first" and row["stage"] == "attendance"
    }
    released_first_counts = {row["track"]: integer(row, "count") for row in attendance_by_edition["2025"]}
    audit.check("first track attendance matches published aggregates", released_first_counts == first_track_counts, f"released={released_first_counts}; source={first_track_counts}")
    later_counts = {row["track"]: integer(row, "count") for row in attendance_by_edition["2026"]}
    expected_later_counts = {
        "Robotics": integer(a["A123"], "value_numeric"),
        "Artificial Intelligence": integer(a["A124"], "value_numeric"),
        "GEO": integer(a["A125"], "value_numeric"),
    }
    audit.check("later track attendance matches weekly headcounts", later_counts == expected_later_counts, f"released={later_counts}; source={expected_later_counts}")
    scale_crosswalk = {
        "SC01": (integer(a["A002"], "value_numeric"), integer(a["A104"], "value_numeric")),
        "SC02": (integer(a["A005"], "value_numeric"), integer(a["A113"], "value_numeric")),
        "SC03": (integer(a["A004"], "value_numeric"), integer(a["A110"], "value_numeric")),
        "SC04": (sum(released_first_counts.values()), sum(later_counts.values())),
        "SC05": (max(released_first_counts.values()), max(later_counts.values())),
        "SC06": (integer(a["A001"], "value_numeric"), integer(a["A132"], "value_numeric")),
    }
    for indicator_id, (first_value, later_value) in scale_crosswalk.items():
        row = scale_by_id[indicator_id]
        audit.check(
            f"program scale source crosswalk {indicator_id}",
            integer(row, "first_value") == first_value and integer(row, "later_value") == later_value,
            f"released={row['first_value']}/{row['later_value']}; source={first_value}/{later_value}",
        )

    track_groups: dict[tuple[str, str], list[dict[str, str]]] = defaultdict(list)
    for row in tracks:
        track_groups[(row["edition"], row["stage"])].append(row)

    stage_admin_crosswalk = {
        ("second", "preferences"): "A107",
        ("second", "admitted assignments"): "A109",
        ("second", "scholarship assignments"): "A111",
    }
    for key, rows in track_groups.items():
        observed = sum(integer(row, "count") for row in rows)
        denominators = {integer(row, "denominator") for row in rows}
        audit.check(f"track denominator internally consistent {key[0]} {key[1]}", len(denominators) == 1, f"declared denominators={sorted(denominators)}")
        if key != ("first", "attendance"):
            audit.check(f"track rows reconstruct declared total {key[0]} {key[1]}", denominators == {observed}, f"row sum={observed}; denominator={sorted(denominators)}")
        if key in stage_admin_crosswalk:
            row_id = stage_admin_crosswalk[key]
            audit.check(f"track total agrees with administrative layer {key[1]}", observed == integer(a[row_id], "value_numeric"), f"track rows={observed}; {row_id}={a[row_id]['value_text']}")

    first_attendance = track_groups[("first", "attendance")]
    audit.check(
        "first-edition track attendance overlap",
        sum(integer(row, "count") for row in first_attendance) > integer(a["A003"], "value_numeric") and all(integer(row, "denominator") == integer(a["A003"], "value_numeric") for row in first_attendance),
        f"overlapping track attendances={sum(integer(row, 'count') for row in first_attendance)}; school attendance={a['A003']['value_text']}",
    )

    combination_groups: dict[tuple[str, str], list[dict[str, str]]] = defaultdict(list)
    for row in track_combinations:
        combination_groups[(row["edition"], row["source_stage"])].append(row)
    expected_combination_groups = {
        ("2025", "entry registrations"): 153,
        ("2026", "applicant preferences"): 444,
        ("2026", "admission"): 280,
        ("2026", "admitted with submitted registration"): 224,
    }
    audit.check(
        "track combination stages",
        set(combination_groups) == set(expected_combination_groups),
        f"observed={sorted(combination_groups)}",
    )
    for key, denominator in expected_combination_groups.items():
        rows = combination_groups[key]
        audit.check(
            f"track combination total {key[0]} {key[1]}",
            len(rows) == 7
            and sum(integer(row, "count") for row in rows) == denominator
            and {integer(row, "denominator") for row in rows} == {denominator},
            f"rows={len(rows)}; sum={sum(integer(row, 'count') for row in rows)}; denominator={denominator}",
        )

    first_entry = {row["path_code"]: integer(row, "count") for row in combination_groups[("2025", "entry registrations")]}
    audit.check(
        "first entry combinations reconstruct track selections",
        first_entry == {"ROB": 22, "DT": 16, "AI": 49, "ROB_DT": 4, "DT_AI": 11, "ROB_AI": 6, "ROB_DT_AI": 45},
        f"observed={first_entry}",
    )
    later_preferences = {row["path_code"]: integer(row, "count") for row in combination_groups[("2026", "applicant preferences")]}
    audit.check(
        "later preference combinations",
        later_preferences == {"ROB": 42, "AI": 68, "GEO": 137, "ROB_AI": 63, "AI_GEO": 35, "ROB_GEO": 3, "ROB_AI_GEO": 96},
        f"observed={later_preferences}",
    )
    later_admission = {row["path_code"]: integer(row, "count") for row in combination_groups[("2026", "admission")]}
    audit.check(
        "later admission combinations",
        later_admission == {"ROB": 62, "AI": 56, "GEO": 78, "ROB_AI": 79, "AI_GEO": 3, "ROB_GEO": 1, "ROB_AI_GEO": 1},
        f"observed={later_admission}",
    )
    later_linked = {row["path_code"]: integer(row, "count") for row in combination_groups[("2026", "admitted with submitted registration")]}
    audit.check(
        "later linked registration combinations",
        later_linked == {"ROB": 46, "AI": 46, "GEO": 64, "ROB_AI": 64, "AI_GEO": 2, "ROB_GEO": 1, "ROB_AI_GEO": 1},
        f"observed={later_linked}",
    )
    first_multi = sum(value for code, value in first_entry.items() if "_" in code)
    first_multi_ai = sum(value for code, value in first_entry.items() if "_" in code and "AI" in code.split("_"))
    later_multi = sum(value for code, value in later_preferences.items() if "_" in code)
    later_multi_ai = sum(value for code, value in later_preferences.items() if "_" in code and "AI" in code.split("_"))
    audit.check("AI central in first entry combinations", (first_multi, first_multi_ai) == (66, 62), f"multi={first_multi}; include AI={first_multi_ai}")
    audit.check("AI central in later preference combinations", (later_multi, later_multi_ai) == (197, 194), f"multi={later_multi}; include AI={later_multi_ai}")

    preference_admission_groups: dict[str, list[dict[str, str]]] = defaultdict(list)
    for row in preference_admission:
        preference_admission_groups[row["preference_path_code"]].append(row)
    audit.check(
        "preference to admission path inventory",
        set(preference_admission_groups) == {"ROB", "AI", "GEO", "ROB_AI", "AI_GEO", "ROB_GEO", "ROB_AI_GEO"},
        f"observed={sorted(preference_admission_groups)}",
    )
    for path_code, rows in preference_admission_groups.items():
        denominators = {integer(row, "preference_path_denominator") for row in rows}
        audit.check(
            f"preference to admission multiplicity {path_code}",
            len(rows) == 4
            and {integer(row, "admitted_track_count") for row in rows} == {0, 1, 2, 3}
            and len(denominators) == 1
            and sum(integer(row, "count") for row in rows) == next(iter(denominators)),
            f"rows={len(rows)}; total={sum(integer(row, 'count') for row in rows)}; denominators={sorted(denominators)}",
        )
    all_three_transition = {
        integer(row, "admitted_track_count"): integer(row, "count")
        for row in preference_admission_groups["ROB_AI_GEO"]
    }
    audit.check(
        "all three preference admission multiplicity",
        all_three_transition == {0: 32, 1: 26, 2: 37, 3: 1},
        f"observed={all_three_transition}",
    )
    audit.check(
        "selection decision unit is applicant by selected track",
        {row["decision_unit"] for row in preference_admission} == {"applicant by selected track"},
        f"observed={sorted({row['decision_unit'] for row in preference_admission})}",
    )

    for row in binary:
        successes = integer(row, "successes")
        n = integer(row, "n")
        audit.check(
            f"later categorical indicator {row['indicator_id']}",
            0 <= successes <= n and n == integer(a["A130"], "value_numeric"),
            f"Yes rows={successes}; complete item rows={n}",
        )

    for row in first_workbook_check:
        implied = integer(row, "published_implied_successes")
        n = integer(row, "n")
        published_percentage = float(row["published_percentage"])
        deposited = integer(row, "deposited_workbook_successes")
        expected_status = "match" if implied == deposited else "unresolved_mismatch"
        audit.check(
            f"first-edition published percentage {row['indicator_id']}",
            round(100 * implied / n, 1) == published_percentage,
            f"implied={implied}/{n}; rounded={100*implied/n:.1f}%; published={published_percentage:.1f}%",
        )
        audit.check(
            f"first-edition deposited-workbook comparison {row['indicator_id']}",
            row["status"] == expected_status,
            f"published-implied={implied}; deposited={deposited}; status={row['status']}",
        )

    for row in likert:
        counts = [integer(row, f"rating_{rating}") for rating in range(1, 6)]
        n = integer(row, "n")
        audit.check(f"five-point total {row['item_id']}", sum(counts) == n == integer(a["A130"], "value_numeric"), f"sum={sum(counts)}; n={n}")
        audit.check(
            f"Likert endpoint status {row['item_id']}",
            row["endpoint_labels_status"] == "verified_from_form_definition"
            and bool(row["lower_anchor"])
            and bool(row["upper_anchor"]),
            f"verified anchors={row['lower_anchor']} to {row['upper_anchor']}",
        )

    categorical_groups: dict[str, list[dict[str, str]]] = defaultdict(list)
    for row in categorical:
        categorical_groups[row["item_id"]].append(row)
    for item_id, rows in categorical_groups.items():
        declared_n = {integer(row, "n") for row in rows}
        observed = sum(integer(row, "count") for row in rows)
        expected_n = integer(a["A130"], "value_numeric")
        audit.check(f"categorical total {item_id}", declared_n == {expected_n} and observed == expected_n, f"counts={observed}; denominators={sorted(declared_n)}")

    attendance_total = sum(integer(row, "count") for row in attendance_patterns)
    multi_track = sum(
        integer(row, "count")
        for row in attendance_patterns
        if sum(row[field] == "true" for field in ("robotics", "artificial_intelligence", "geo")) > 1
    )
    attendance_by_track = {
        field: sum(integer(row, "count") for row in attendance_patterns if row[field] == "true")
        for field in ("robotics", "artificial_intelligence", "geo")
    }
    audit.check("questionnaire attendance-pattern total", attendance_total == integer(a["A130"], "value_numeric"), f"sum={attendance_total}; questionnaire rows={a['A130']['value_text']}")
    audit.check("multi-track response rows derived", 0 <= multi_track <= attendance_total, f"observed={multi_track}")
    audit.check(
        "questionnaire track-selection counts are overlapping",
        sum(attendance_by_track.values()) >= attendance_total,
        f"observed={attendance_by_track}",
    )

    for row in country_summary:
        raw = integer(row, "raw_distinct_labels")
        canonical = integer(row, "canonical_distinct_countries")
        audit.check(f"country normalization {row['stage']}", raw - canonical == 1, f"raw={raw}; canonical={canonical}")

    profile_groups: dict[tuple[str, str], list[dict[str, str]]] = defaultdict(list)
    for row in profiles:
        profile_groups[(row["edition"], row["stage"])].append(row)
    profile_crosswalk = {
        ("first", "actual attendees"): "A003",
        ("second", "decision dataset"): "A104",
        ("second", "admitted IDs"): "A108",
        ("second", "questionnaire response rows"): "A130",
    }
    for key, rows in profile_groups.items():
        row_id = profile_crosswalk[key]
        expected = integer(a[row_id], "value_numeric")
        observed = sum(integer(row, "count") for row in rows)
        denominators = {integer(row, "denominator") for row in rows}
        audit.check(f"profile total {key[0]} {key[1]}", observed == expected and denominators == {expected}, f"sum={observed}; denominators={sorted(denominators)}")

    for row in demographics:
        valid = integer(row, "valid_binary_gender_records")
        observed = integer(row, "female") + integer(row, "male")
        audit.check(f"registration gender {row['pathway']}", observed == valid, f"female + male={observed}; valid={valid}")

    agenda_practical = {row["track"]: integer(row, "practical_slots") for row in agenda}
    slot_counts = Counter(row["track"] for row in slots)
    audit.check("agenda practical total", sum(agenda_practical.values()) == 11, f"agenda summary total={sum(agenda_practical.values())}")
    audit.check("practical slot row total", len(slots) == 11, f"slot rows={len(slots)}")
    audit.check("distinct practical-format total", len({row["format_id"] for row in slots}) == 8, f"distinct formats={len({row['format_id'] for row in slots})}")
    audit.check("practical slots by track", dict(slot_counts) == agenda_practical, f"rows={dict(slot_counts)}; summary={agenda_practical}")
    rover_slots = sum(row["format_id"] == "rover_systems_lab" for row in slots)
    audit.check("Rover format repeated slots", rover_slots == 4, f"observed={rover_slots}")
    slot_ids = {row["slot_id"] for row in slots}
    coding_ids = {row["slot_id"] for row in agenda_coding}
    coding_titles = {row["slot_id"]: row["sanitized_title"] for row in agenda_coding}
    slot_titles = {row["slot_id"]: row["official_title_deidentified"] for row in slots}
    audit.check("agenda author-coding row coverage", coding_ids == slot_ids, f"slot IDs={len(slot_ids)}; coding IDs={len(coding_ids)}")
    audit.check("agenda author-coding title consistency", coding_titles == slot_titles, "sanitized titles agree across the two curated tables")
    audit.check(
        "agenda coding provenance disclosure",
        {row["source_custody_id"] for row in agenda_coding} == {"S09"}
        and {row["verification_status"] for row in agenda_coding} == {"author_curated_not_publicly_source_verified"},
        "all rows are explicitly marked author-curated and not publicly source-verified",
    )

    learning_ids = [row["stage_id"] for row in sorted(learning_content, key=lambda row: integer(row, "stage_order"))]
    audit.check(
        "documented transition stage coverage",
        learning_ids == ["initial_program", "initial_evidence", "documented_decisions", "later_program", "later_evidence", "prospective_protocol", "inference_boundary"],
        f"observed={learning_ids}",
    )
    learning_text = " ".join(row["stage_title"] + " " + row["content"] for row in learning_content)
    audit.check("prospective protocol is not a completed cycle", "cycle" not in learning_text.casefold() and next(row for row in learning_content if row["stage_id"] == "prospective_protocol")["evidence_role"] == "protocol_stage", "one documented transition ends in a prospective protocol")
    audit.check("transition inference boundary", "causal effects" in learning_text and "cost effectiveness" in learning_text, "unsupported outcomes are explicit")

    objective_ids = [row["objective_id"] for row in objectives]
    audit.check("program logic inventory", objective_ids == ["O1", "O2", "O3", "O4", "EP1", "EP2", "S1"], f"observed={objective_ids}")
    status_by_objective = {row["objective_id"]: row["status"] for row in objectives}
    audit.check(
        "goal status model",
        status_by_objective == {"O1": "founding_goal", "O2": "founding_goal", "O3": "founding_goal", "O4": "founding_goal", "EP1": "emergent_priority", "EP2": "emergent_priority", "S1": "support_condition"},
        f"observed={status_by_objective}",
    )
    audit.check("responsible innovation remains O4", "Responsible innovation" in next(row for row in objectives if row["objective_id"] == "O4")["title"], "O4 remains a founding goal")
    audit.check("mission practice is an emergent priority", "Mission oriented" in next(row for row in objectives if row["objective_id"] == "EP1")["title"], "EP1 is analytically distinct from the founding objectives")
    audit.check("face to face dialogue and networking is an emergent priority", "face to face" in next(row for row in objectives if row["objective_id"] == "EP2")["title"], "EP2 is analytically distinct from the founding objectives")
    audit.check("participation support is S1", next(row for row in objectives if row["objective_id"] == "S1")["status"] == "support_condition", "S1 is not an educational outcome")

    decision_ids = [row["decision_id"] for row in decisions]
    audit.check("design decision inventory", decision_ids == [f"D{index}" for index in range(1, 12)], f"observed={decision_ids}")
    decision_statuses = Counter(row["evidence_status"] for row in decisions)
    audit.check(
        "design decision evidence statuses",
        decision_statuses == {"founding_documented": 5, "adaptation_pathway_documented": 3, "mechanism_documented_partial_history": 3},
        f"observed={dict(decision_statuses)}",
    )
    objective_id_set = set(objective_ids)
    rq_id_set = {"RQ1", "RQ2a", "RQ2b", "RQ3"}
    decision_objectives = {
        row["decision_id"]: {code.strip() for code in row["objective_codes"].split(",")}
        for row in decisions
    }
    decision_rqs = {
        row["decision_id"]: {code.strip() for code in row["rq_links"].split(",")}
        for row in decisions
    }
    audit.check(
        "decision objective links valid",
        all(codes and codes.issubset(objective_id_set) for codes in decision_objectives.values()),
        f"observed={decision_objectives}",
    )
    audit.check(
        "decision RQ links valid",
        all(codes and codes.issubset(rq_id_set) for codes in decision_rqs.values()),
        f"observed={decision_rqs}",
    )
    objective_decisions = {
        row["objective_id"]: {code.strip() for code in row["decision_ids"].split(",")}
        for row in objectives
    }
    reverse_decision_links = {
        objective_id: {decision_id for decision_id, codes in decision_objectives.items() if objective_id in codes}
        for objective_id in objective_ids
    }
    audit.check(
        "objective decision mapping is bidirectional",
        objective_decisions == reverse_decision_links,
        f"objective rows={objective_decisions}; decision rows={reverse_decision_links}",
    )
    audit.check(
        "every objective has multiple design decisions",
        all(len(codes) >= 2 for codes in objective_decisions.values()),
        f"coverage={ {key: len(value) for key, value in objective_decisions.items()} }",
    )
    expected_rq_decisions = {
        "RQ1": {"D1", "D2", "D3", "D4", "D5", "D6", "D7", "D10", "D11"},
        "RQ2a": {"D1", "D2", "D3", "D6", "D7", "D8", "D9", "D10"},
        "RQ2b": set(),
        "RQ3": {"D3", "D4", "D5", "D9", "D11"},
    }
    observed_rq_decisions = {
        rq: {decision_id for decision_id, links in decision_rqs.items() if rq in links}
        for rq in rq_id_set
    }
    audit.check("decision coverage by RQ", observed_rq_decisions == expected_rq_decisions, f"observed={observed_rq_decisions}")

    protocol_ids = [row["protocol_id"] for row in protocol]
    protocol_codes = {
        code.strip()
        for row in protocol
        for code in row["objective_codes"].split(",")
    }
    audit.check("prospective evaluation protocol inventory", protocol_ids == [f"PE{index:02d}" for index in range(1, 8)], f"observed={protocol_ids}")
    protocol_required_fields = (
        "proposition_ids", "evaluation_question", "claim_to_test", "unit_and_timing",
        "primary_outcome", "sampling_and_comparator", "measurement_and_quality",
        "analysis", "decision_rule", "missing_data_rule", "interpretation_rule",
    )
    audit.check(
        "empirical evaluation plan fields complete",
        all(all(row[field] for field in protocol_required_fields) for row in protocol),
        "all prospective tests define outcomes comparators measurement analysis decision missing data and interpretation rules",
    )
    audit.check("prospective protocol goal links valid", protocol_codes.issubset(objective_id_set), f"observed={sorted(protocol_codes)}")

    speaker_groups: dict[tuple[str, str], list[dict[str, str]]] = defaultdict(list)
    for row in speaker_composition:
        speaker_groups[(row["edition"], row["documentary_stage"])].append(row)
    speaker_admin_crosswalk = {
        ("initial", "contributing speakers"): "A001",
        ("later", "accepted-speaker records"): "A132",
    }
    for key, rows in speaker_groups.items():
        expected = integer(a[speaker_admin_crosswalk[key]], "value_numeric")
        observed = sum(integer(row, "count") for row in rows)
        denominators = {integer(row, "denominator") for row in rows}
        categories = {row["category"] for row in rows}
        audit.check(f"speaker-record composition {key[0]}", observed == expected and denominators == {expected} and categories == {"Industry-classified", "Other sectors"}, f"sum={observed}; denominators={sorted(denominators)}; categories={sorted(categories)}")

    comparator_keys = {row["citation_key"] for row in comparators}
    audit.check(
        "longitudinal comparator inventory",
        len(comparators) == 6 and len(comparator_keys) == 6,
        f"rows={len(comparators)}; unique citation keys={len(comparator_keys)}",
    )

    status_counts = Counter(row["evidence_status"] for row in traceability)
    driver_ids = {row["driver_id"] for row in rq1_themes}
    traceability_ids = {row["driver_id"] for row in traceability}
    audit.check("RQ1 driver-frame inventory", len(rq1_themes) == 3 and driver_ids == {f"D{index:02d}" for index in range(1, 4)}, f"rows={len(rq1_themes)}; IDs={sorted(driver_ids)}")
    audit.check("RQ1 driver-to-traceability coverage", traceability_ids == driver_ids, f"driver IDs={len(driver_ids)}; traceability IDs={len(traceability_ids)}")
    audit.check(
        "RQ1 decision pathway mapping",
        {row["decision_id"] for row in traceability} == {"D6", "D7", "D11"}
        and {row["decision_id"] for row in decisions if row["evidence_status"] == "adaptation_pathway_documented"} == {"D6", "D7", "D11"},
        f"traceability={sorted(row['decision_id'] for row in traceability)}",
    )
    audit.check("RQ1 source disclosure", all(row["source_status"] and row["source_locator"] and row["sampling_rule"] for row in rq1_themes), "all drivers declare source status locator and sampling rule")
    audit.check("RQ1 main traceability threshold", status_counts == {"documented": 3}, f"statuses={dict(status_counts)}")
    evidence_ids = [row["rq"] for row in evidence]
    audit.check("three-RQ evidence inventory with RQ2 analytical parts", evidence_ids == ["RQ1", "RQ2a", "RQ2b", "RQ3"], f"observed={evidence_ids}")
    evidence_questions = {row["rq"]: row["question"] for row in evidence}
    audit.check("exact manuscript RQ wording in evidence matrix", evidence_questions == EXPECTED_RQS, f"observed={evidence_questions}")
    audit.check("exact manuscript RQ wording in public study design", all(question in rq_document for question in EXPECTED_RQS.values()), "all questions and RQ2 analytical parts occur verbatim")
    rq_design_text = (rq_document + " " + " ".join(row["analytical_target"] + " " + row["inference_boundary"] for row in evidence)).casefold()
    audit.check("no coherence validation claim", "design coherence" not in rq_design_text and "coherent learning environment" not in rq_design_text, "RQ2a is descriptive and RQ2b reports reactions without validating author constructed alignment")

    allowed_questionnaire_fields = {
        "item_id", "construct", "response_type", "category", "count", "n",
        "source_status", "evidence_role", "interpretation_boundary",
    }
    for edition, rows, expected_n in (
        ("2025", questionnaire_2025, 77),
        ("2026", questionnaire_2026, 82),
    ):
        groups: dict[str, list[dict[str, str]]] = defaultdict(list)
        for row in rows:
            groups[row["item_id"]].append(row)
        audit.check(
            f"{edition} questionnaire item coverage",
            set(groups) == {f"Q{index:02d}" for index in range(1, 14)},
            f"items={sorted(groups)}",
        )
        audit.check(
            f"{edition} questionnaire schema is descriptive only",
            all(set(row) == allowed_questionnaire_fields for row in rows)
            and {row["evidence_role"] for row in rows} == {"edition_specific_descriptive_count"},
            "only edition-specific category counts and boundaries are released",
        )
        for item_id, item_rows in groups.items():
            observed = sum(integer(row, "count") for row in item_rows)
            denominators = {integer(row, "n") for row in item_rows}
            audit.check(
                f"{edition} questionnaire total {item_id}",
                observed == expected_n and denominators == {expected_n},
                f"counts={observed}; denominators={sorted(denominators)}",
            )

    audit.check(
        "harmonized questionnaire summary coverage",
        {row["item_id"] for row in questionnaire_common} == {f"Q{index:02d}" for index in range(1, 14)},
        f"items={sorted(row['item_id'] for row in questionnaire_common)}",
    )
    for row in questionnaire_common:
        first_successes = integer(row, "first_successes")
        later_successes = integer(row, "later_successes")
        audit.check(
            f"harmonized questionnaire summary {row['item_id']}",
            integer(row, "first_n") == 77
            and integer(row, "later_n") == 82
            and row["response_rule"] in {"upper categories 4 or 5", "Yes"}
            and float(row["first_percent"]) == round(100 * first_successes / 77, 1)
            and float(row["later_percent"]) == round(100 * later_successes / 82, 1),
            f"2025={first_successes}/77; 2026={later_successes}/82",
        )

    open_inventory = {row["edition"]: row for row in open_comment_inventory}
    audit.check(
        "edition specific open comment inventory",
        integer(open_inventory["2025"], "nonempty_comment_cells") == 162
        and integer(open_inventory["2025"], "response_rows_with_comment") == 38
        and integer(open_inventory["2026"], "nonempty_comment_cells") == 238
        and integer(open_inventory["2026"], "response_rows_with_comment") == 50,
        "2025=162 cells/38 forms; 2026=238 cells/50 forms",
    )

    access = {row["metric_id"]: row for row in access_process}
    expected_access_values = {
        "AP01": 444, "AP02": 280, "AP03": 196, "AP04": 83, "AP05": 1,
        "AP06": 357, "AP07": 220, "AP08": 244, "AP09": 133, "AP10": 128,
        "AP11": 207, "AP12": 207, "AP13": 59, "AP14": 59, "AP15": 224,
        "AP16": 56, "AP17": 25,
    }
    audit.check("access process metric inventory", set(access) == set(expected_access_values), f"observed={sorted(access)}")
    audit.check(
        "access process derived values",
        all(integer(access[metric_id], "value") == expected for metric_id, expected in expected_access_values.items()),
        "all access process values match the restricted source to aggregate assertions",
    )
    audit.check(
        "admission multiplicity reconstruction",
        integer(access["AP03"], "value") + integer(access["AP04"], "value") + integer(access["AP05"], "value") == integer(access["AP02"], "value"),
        "one, two, and three track admission groups reconstruct distinct admitted identifiers",
    )
    audit.check(
        "registration linkage accounting",
        integer(access["AP15"], "value") + integer(access["AP16"], "value") == integer(access["AP02"], "value"),
        "exactly linked and unresolved admitted identifiers reconstruct the admitted set",
    )

    capacity = {row["track"]: row for row in capacity_pressure}
    expected_capacity = {"Robotics": (143, 9), "AI": (139, 52), "GEO": (83, 16)}
    audit.check("capacity pressure track inventory", set(capacity) == set(expected_capacity), f"tracks={sorted(capacity)}")
    audit.check(
        "capacity pressure values",
        all(
            integer(capacity[track], "admitted_assignments") == values[0]
            and integer(capacity[track], "eligible_not_admitted_assignments") == values[1]
            and integer(capacity[track], "documented_operating_limit") == 140
            for track, values in expected_capacity.items()
        ),
        "track assignments and eligible counts agree with the released status sheet",
    )

    selection_groups: dict[str, list[dict[str, str]]] = defaultdict(list)
    for row in scholarship_model:
        selection_groups[row["applicant_group"]].append(row)
    audit.check(
        "scholarship scoring totals",
        sum(integer(row, "maximum_points") for row in selection_groups["Academic"] if row["maximum_points"]) == 100
        and sum(integer(row, "maximum_points") for row in selection_groups["Practitioner"] if row["maximum_points"]) == 100,
        "academic and practitioner weighted fields each sum to 100 points",
    )
    gender_row = next(row for row in scholarship_model if row["criterion"] == "Gender equality and inclusiveness")
    audit.check(
        "published inclusion criterion lacks a dedicated weighted field",
        gender_row["published_criteria_document"] == "yes" and gender_row["operational_workbook_field"] == "no dedicated weighted field",
        "criterion is disclosed as a traceability gap rather than an evaluated equity mechanism",
    )

    allocation_rule = {row["component_id"]: row for row in scholarship_rule}
    audit.check(
        "scholarship allocation component inventory",
        set(allocation_rule) == {"SA01", "SA02", "SA03"},
        f"observed={sorted(allocation_rule)}",
    )
    audit.check(
        "country indexed travel allowance range",
        integer(allocation_rule["SA01"], "amount_min_eur") == 100
        and integer(allocation_rule["SA01"], "amount_max_eur") == 1300
        and integer(allocation_rule["SA01"], "lookup_entries") == 70,
        "70 country lookup entries span EUR 100 to EUR 1,300",
    )
    audit.check(
        "common weekly accommodation allowance",
        integer(allocation_rule["SA02"], "amount_min_eur") == 350
        and integer(allocation_rule["SA02"], "amount_max_eur") == 350
        and "each awarded track week" in allocation_rule["SA02"]["application_frequency"],
        "the common weekly component is EUR 350 per awarded week",
    )
    audit.check(
        "travel allowance applied once across awarded tracks",
        "Once per scholarship recipient" in allocation_rule["SA01"]["application_frequency"]
        and "Travel allowance included once" in allocation_rule["SA03"]["calculation_basis"],
        "the country indexed travel component is not repeated for additional awarded weeks",
    )
    audit.check(
        "scholarship allocation evidence trace",
        {row["source_ids"] for row in scholarship_rule} == {"S03", "S03, OR13"},
        "workbook formulas support all rows and organizer interpretation supports the cost basis",
    )
    audit.check(
        "scholarship allocation inference boundary",
        all("not" in row["inference_boundary"].casefold() for row in scholarship_rule)
        and "equal opportunity" in allocation_rule["SA01"]["inference_boundary"],
        "the released rule does not claim cost sufficiency equal opportunity payment or attendance",
    )

    speakers_track = {(row["track"], row["sector_classification"]): integer(row, "accepted_records") for row in speaker_by_track}
    expected_speakers_track = {
        ("Robotics", "Industry classified"): 6, ("Robotics", "Other sectors"): 11,
        ("AI", "Industry classified"): 19, ("AI", "Other sectors"): 3,
        ("GEO", "Industry classified"): 0, ("GEO", "Other sectors"): 16,
    }
    audit.check("speaker composition by later track", speakers_track == expected_speakers_track, f"observed={speakers_track}")

    proposition_ids = [row["proposition_id"] for row in propositions]
    audit.check("proposition and evaluation principle inventory", proposition_ids == ["P1", "P2", "P3", "P4", "P5", "E1"], f"observed={proposition_ids}")
    audit.check(
        "proposition outcomes remain explicit and unmeasured",
        all(row["outcome_not_measured"] for row in propositions),
        "every candidate relationship separates observed basis from its unmeasured outcome",
    )
    empirical_proposition_ids = {proposition_id for proposition_id in proposition_ids if proposition_id.startswith("P")}
    protocol_links = {
        code.strip()
        for row in protocol
        for code in row["proposition_ids"].split(",")
    }
    audit.check("prospective protocol proposition links valid", protocol_links == empirical_proposition_ids, f"observed={sorted(protocol_links)}")
    audit.check("every empirical proposition has a prospective test", all(row["prospective_protocol_ids"] for row in propositions if row["proposition_id"].startswith("P")), "P1 through P5 define one or more protocol IDs")
    evaluation_principle = next(row for row in propositions if row["proposition_id"] == "E1")
    audit.check("evaluation principle is separate from empirical propositions", evaluation_principle["prospective_protocol_ids"] == "Applies to PE01 through PE07" and "methodological" in evaluation_principle["claim_boundary"], "E1 applies to every protocol row and is not an empirical proposition")

    output_ids = [row["output_id"] for row in output_traceability]
    audit.check("paper output traceability inventory", output_ids == ["T01", "F01", "T02", "T03", "T04", "T05", "F02", "F03", "T06", "T07", "T08", "T09", "T10", "TA11", "TA12", "TA13", "P01"], f"observed={output_ids}")
    audit.check("paper output coverage status", {row["coverage_status"] for row in output_traceability} == {"complete"}, "every listed paper artifact has canonical input calculation output and boundary")
    audit.check("paper output traceability fields complete", all(row["canonical_input"] and row["calculation_or_rule"] and row["generated_output"] and row["verification_boundary"] for row in output_traceability), "all traceability rows are complete")

    synthesis_text = " ".join(row["scope_and_formalization"] for row in explanatory_synthesis)
    synthesis_links = set(re.findall(r"\b(?:P[1-5]|E1)\b", synthesis_text))
    audit.check("design lesson synthesis inventory", [row["lesson_id"] for row in explanatory_synthesis] == ["DL1", "DL2", "DL3", "DL4", "DL5"], f"rows={len(explanatory_synthesis)}")
    audit.check("design lesson synthesis links valid", synthesis_links == {"P1", "P2", "P3", "P4", "P5", "E1"}, f"observed={sorted(synthesis_links)}")

    curricular_keys = [(row["edition"], row["program_area"]) for row in curricular_connections]
    audit.check("curricular connection inventory", curricular_keys == [("2025", "Robotics"), ("2025", "Digital Twins"), ("2025", "AI"), ("2026", "Robotics"), ("2026", "AI"), ("2026", "GEO")], f"observed={curricular_keys}")
    curricular_text = " ".join(row["perspectives_present"] + " " + row["examples_in_records"] for row in curricular_connections)
    audit.check("curricular interdisciplinary coverage", all(term in curricular_text for term in ("Social Science", "Physics", "Astrophysics", "Software Engineering", "Digital Twins")), "computing Social Science Physics Astrophysics and Digital Twins are represented")

    metadata = {row["metadata_id"]: row for row in corpus_metadata}
    audit.check("corpus metadata inventory", set(metadata) == {f"CM{index:02d}" for index in range(1, 15)}, f"observed={sorted(metadata)}")
    audit.check("questionnaire corpus metadata", metadata["CM10"]["value_text"] == "34" and metadata["CM11"]["value_text"] == "2026 05 08" and metadata["CM12"]["value_text"] == "2026 06 03", "34 variables and retained collection date range are released")
    audit.check("multiple track questionnaire metadata", metadata["CM03"]["value_text"] == "34" and metadata["CM13"]["value_text"] == "22", "multiple track form counts are released for both editions")
    audit.check("planning and role metadata", metadata["CM04"]["value_text"] == "2025 09 11" and metadata["CM05"]["value_text"] == "2025 10 10" and metadata["CM06"]["value_text"] == "2025 10 22" and metadata["CM14"]["value_text"] == "35", "three planning dates and role register breadth are released")

    detailed_groups: dict[tuple[str, str], int] = defaultdict(int)
    for row in speaker_detailed:
        detailed_groups[(row["edition"], row["documentary_stage"])] += integer(row, "count")
    audit.check("detailed speaker composition totals", detailed_groups == {("2025", "contributing speakers"): 58, ("2026", "accepted speaker records"): 55}, f"observed={dict(detailed_groups)}")

    networking_ids = [row["activity_id"] for row in networking_design]
    audit.check("networking activity design inventory", networking_ids == ["N01", "N02"], f"observed={networking_ids}")
    networking_text = " ".join(row["implemented_feature"] + " " + row["participant_interaction"] for row in networking_design)
    audit.check("networking activity features documented", all(term in networking_text for term in ("privacy aware", "QR access", "individual and table rankings", "reference solution")), "prearrival communication and app supported social interaction features are represented")

    communication_ids = [row["mechanism_id"] for row in communication_use]
    audit.check("communication evidence inventory", communication_ids == ["C01", "C02", "C03"], f"observed={communication_ids}")
    communication_text = " ".join(row["observable_evidence"] for row in communication_use)
    audit.check(
        "reported communication use categories",
        all(term in communication_text for term in ("travel", "accommodation", "shared hotel", "logistical notices", "speaker materials", "slides")),
        "participant prearrival coordination and organizer information and material distribution are represented",
    )
    communication_boundaries = " ".join(row["inference_boundary"].lower() for row in communication_use)
    audit.check(
        "communication use claims bounded",
        all(term in communication_boundaries for term in ("message volume", "active-user reach", "continued contact", "readership", "durable relationships")),
        "reported use categories remain distinct from activity reach relationship formation and persistence",
    )

    rover_ids = [row["element_id"] for row in rover_design]
    audit.check("Rover Systems Lab design inventory", rover_ids == [f"RSL{index:02d}" for index in range(1, 8)], f"observed={rover_ids}")
    rover_text = " ".join(row["documented_2026_realization"] + " " + row["educational_focus"] + " " + row["reuse_through_academy"] for row in rover_design)
    audit.check("Rover Systems Lab design coverage", all(term in rover_text for term in ("simplified lunar", "simulated rover", "six macro groups", "collective objective", "nominal points", "GSSI Robotics Laboratory", "design review", "four ninety minute agenda slots", "secondary school outreach")), "mission framing intergroup cooperation platform sequence organization and reuse are represented")

    expertise_by_track = {row["track_code"]: row for row in expertise_fields}
    audit.check("application expertise field inventory", set(expertise_by_track) == {"ROB", "AI", "GEO"}, f"observed={sorted(expertise_by_track)}")
    audit.check(
        "application expertise selected counts",
        {code: integer(row, "selected_applicants") for code, row in expertise_by_track.items()} == {"ROB": 204, "AI": 262, "GEO": 271},
        "track specific prompt denominators match declared preferences",
    )
    audit.check(
        "application expertise non placeholder counts",
        {code: integer(row, "non_placeholder_skill_entries") for code, row in expertise_by_track.items()} == {"ROB": 183, "AI": 227, "GEO": 270},
        "aggregate completeness is reproduced from the restricted workbook",
    )
    audit.check(
        "application expertise missing counts",
        all(integer(row, "selected_applicants") == integer(row, "non_placeholder_skill_entries") + integer(row, "missing_or_placeholder_entries") for row in expertise_fields),
        "selected entries equal non placeholder plus missing or placeholder entries",
    )
    audit.check(
        "Robotics group formation claim bounded",
        "six macro groups" in expertise_by_track["ROB"]["reported_grouping_use"] and "no assignment matrix" in expertise_by_track["ROB"]["inference_boundary"].lower(),
        "group composition is reported and its missing reconstruction artifact is explicit",
    )

    schema_ids = [row["source_id"] for row in source_schemas]
    audit.check("source schema inventory", schema_ids == [f"S{index:02d}" for index in range(1, 18)], f"observed={schema_ids}")
    audit.check("source schema fields complete", all(row["source_unit"] and row["source_schema_summary"] and row["record_count_or_scope"] and row["checksum_or_public_identifier"] and row["public_outputs"] and row["transformation_summary"] and row["withheld_fields"] and row["release_status"] for row in source_schemas), "all source schemas define scope identifier output transformation and withheld fields")

    node_ids = [row["node_id"] for row in transition_nodes]
    edge_pairs = [(row["source_node"], row["target_node"]) for row in transition_edges]
    audit.check("program transition node inventory", node_ids == ["initial", "evaluation", "redesign", "second", "current", "future"], f"observed={node_ids}")
    audit.check("program transition edge inventory", edge_pairs == [("initial", "evaluation"), ("evaluation", "redesign"), ("redesign", "second"), ("second", "current"), ("current", "future")], f"observed={edge_pairs}")
    audit.check("program transition edge endpoints valid", all(source in set(node_ids) and target in set(node_ids) for source, target in edge_pairs), "all edge endpoints resolve to released nodes")
    audit.check("program transition prospective boundary", [row["edge_type"] for row in transition_edges] == ["documented", "documented", "documented", "documented", "prospective"], "only the final relation is prospective")

    audit.check("paper program logic row inventory", [row["objective_id"] for row in paper_objectives] == objective_ids, "paper table covers O1 to O4, EP1, EP2, and S1")
    paper_corpus_rqs = {
        code.strip()
        for row in paper_corpus
        for code in row["rq"].split(",")
    }
    audit.check("paper corpus RQ links valid", paper_corpus_rqs == rq_id_set, f"observed={sorted(paper_corpus_rqs)}")
    audit.check("paper corpus source inventory", len(paper_corpus) == 6 and {row["edition"] for row in paper_corpus} == {"2025", "2026", "Both"}, f"rows={len(paper_corpus)}")
    audit.check("paper participation support inventory", len(paper_participation) == 9 and [row["edition"] for row in paper_participation].count("2025") == 2 and [row["edition"] for row in paper_participation].count("2026") == 7, f"rows={len(paper_participation)}")

    public_data_files = list(DATA.glob("*.csv")) + list(DESIGN.glob("*.csv")) + list(INSTRUMENTS.glob("*.csv")) + list(GOVERNANCE.glob("*.csv"))
    email_pattern = re.compile(r"[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}", re.IGNORECASE)
    absolute_path_pattern = re.compile(r"/(Users|home)/[^,\s\"]+")
    email_hits: list[str] = []
    path_hits: list[str] = []
    for path in public_data_files:
        text = path.read_text(encoding="utf-8")
        if email_pattern.search(text):
            email_hits.append(path.name)
        if absolute_path_pattern.search(text):
            path_hits.append(path.name)
    audit.check("no email addresses in public canonical tables", not email_hits, f"files={email_hits}")
    audit.check("no local absolute paths in public canonical tables", not path_hits, f"files={path_hits}")

    dictionary_files = {row["file"] for row in data_dictionary}
    known_csvs = {path.name for path in public_data_files}
    audit.check("data-dictionary file references exist", dictionary_files.issubset(known_csvs), f"missing={sorted(dictionary_files - known_csvs)}")
    dictionary_failures: list[str] = []
    for row in data_dictionary:
        candidates = [DATA / row["file"], DESIGN / row["file"], INSTRUMENTS / row["file"], GOVERNANCE / row["file"]]
        path = next((candidate for candidate in candidates if candidate.exists()), None)
        if path is None:
            dictionary_failures.append(f"{row['file']}:{row['field']} (file missing)")
            continue
        with path.open(newline="", encoding="utf-8") as handle:
            headers = set(next(csv.reader(handle)))
        if not dictionary_field_matches(row["field"], headers):
            dictionary_failures.append(f"{row['file']}:{row['field']}")
    audit.check("data-dictionary fields match released schemas", not dictionary_failures, f"invalid={dictionary_failures}")
    return audit


def format_indicator(successes: int, n: int) -> str:
    return f"{successes}/{n} ({pct(successes/n)}\\%)"


def generate_respondent_table(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table}[!t]",
        r"\centering",
        r"\scriptsize",
        r"\caption{Descriptive indicators across 82 second-edition returned questionnaires. Unique respondents and the invitation denominator are unknown; the values describe this returned-form set only.}",
        r"\label{tab:respondent-indicators-generated}",
        r"\begin{tabularx}{\columnwidth}{p{0.43\columnwidth} p{0.25\columnwidth} Y}",
        r"\toprule",
        r"\textbf{Indicator} & \textbf{Observed response} & \textbf{Response rule} \\",
        r"\midrule",
    ]
    for index, row in enumerate(rows):
        line = " & ".join(
            [
                tex_escape(row["indicator"]),
                format_indicator(integer(row, "successes"), integer(row, "n")),
                tex_escape(row["response_rule"]),
            ]
        ) + r" \\"
        lines.append(line)
        if index < len(rows) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table}", ""])
    return "\n".join(lines)


def generate_evidence_table(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]",
        r"\centering",
        r"\footnotesize",
        r"\caption{Evidence sources and inference boundaries for the three research questions.}",
        r"\label{tab:evidence-by-rq-generated}",
        r"\begin{tabularx}{\textwidth}{p{0.06\textwidth} p{0.24\textwidth} Y Y}",
        r"\toprule",
        r"\textbf{RQ} & \textbf{Analytical target} & \textbf{Direct evidence} & \textbf{Claim boundary} \\",
        r"\midrule",
    ]
    for index, row in enumerate(rows):
        lines.append(" & ".join(tex_escape(row[key]) for key in ("rq", "analytical_target", "direct_evidence", "inference_boundary")) + r" \\")
        if index < len(rows) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_traceability_table(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]",
        r"\centering",
        r"\scriptsize",
        r"\setlength{\tabcolsep}{3pt}",
        r"\renewcommand{\arraystretch}{1.05}",
        r"\caption{Evidence chains for the three changes traced from 2025 to 2026.}",
        r"\label{tab:objective-evolution}",
        r"\begin{tabularx}{\textwidth}{p{0.22\textwidth} p{0.27\textwidth} p{0.24\textwidth} Y}",
        r"\toprule",
        r"\textbf{Prior documented basis} & \textbf{Alternatives and decision} & \textbf{Implemented change} & \textbf{Observed signal and boundary} \\",
        r"\midrule",
    ]
    for index, row in enumerate(rows):
        values = [
            row["paper_prior_documented_basis"],
            row["paper_alternatives_and_decision"],
            row["paper_implemented_change"],
            row["paper_observed_signal_and_boundary"],
        ]
        lines.append(" & ".join(paper_track_terms(value) for value in values) + r" \\")
        if index < len(rows) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_comparator_table(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\scriptsize",
        r"\setlength{\tabcolsep}{3pt}",
        r"\caption{Longitudinal education comparators and the empirical distinction of this study.}",
        r"\label{tab:longitudinal-comparators-generated}",
        r"\begin{tabularx}{\textwidth}{p{0.17\textwidth} p{0.23\textwidth} Y Y}",
        r"\toprule",
        r"\textbf{Study} & \textbf{Longitudinal object} & \textbf{Primary evidence} & \textbf{Contrast with this case} \\",
        r"\midrule",
    ]
    for index, row in enumerate(rows):
        values = [
            rf"\citet{{{row['citation_key']}}}",
            tex_escape(row["longitudinal_object"]),
            tex_escape(row["primary_evidence"]),
            tex_escape(row["contrast_with_case"]),
        ]
        lines.append(" & ".join(values) + r" \\")
        if index < len(rows) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_learning_tikz(rows: list[dict[str, str]]) -> str:
    by_id = {row["node_id"]: row for row in rows}
    initial = by_id["initial"]
    evaluation = by_id["evaluation"]
    redesign = by_id["redesign"]
    second = by_id["second"]
    current = by_id["current"]
    future = by_id["future"]

    return rf"""\begin{{figure*}}[!t]
\centering
\begin{{tikzpicture}}[
  stage/.style={{draw=spblue!80, fill=spblue!5, rounded corners=2pt,
    line width=.75pt, align=left, text width=4.65cm,
    minimum height=2.12cm, inner sep=5pt, font=\scriptsize}},
  decision/.style={{draw=spgreen!80!black, fill=spgreen!6, rounded corners=2pt,
    line width=.75pt, align=left, text width=4.65cm,
    minimum height=2.12cm, inner sep=5pt, font=\scriptsize}},
  evidence/.style={{draw=sporange!85!black, fill=sporange!6, rounded corners=2pt,
    line width=.75pt, align=left, text width=4.65cm,
    minimum height=2.12cm, inner sep=5pt, font=\scriptsize}},
  protocol/.style={{draw=spgray!85, fill=spgray!5, rounded corners=2pt,
    line width=.75pt, align=left, text width=4.65cm,
    minimum height=2.12cm, inner sep=5pt, font=\scriptsize}},
  flow/.style={{-{{Latex[length=2.5mm,width=1.7mm]}}, draw=spgray!95,
    line width=1.05pt}},
  prospective/.style={{-{{Latex[length=2.5mm,width=1.7mm]}}, draw=spgreen!80!black,
    dashed, line width=1.05pt}}
]
\node[stage] (initial) {{\textbf{{{tex_escape(initial['title'])}}}\\[3pt]
{transition_content('initial', initial['content'])}}};

\node[evidence, right=0.65cm of initial] (evaluation) {{\textbf{{{tex_escape(evaluation['title'])}}}\\[3pt]
{tex_escape(evaluation['content'])}}};

\node[decision, right=0.65cm of evaluation] (redesign) {{\textbf{{{tex_escape(redesign['title'])}}}\\[3pt]
{transition_content('redesign', redesign['content'])}}};

\node[stage, below=0.60cm of redesign] (second) {{\textbf{{{tex_escape(second['title'])}}}\\[3pt]
{transition_content('second', second['content'])}}};

\node[evidence, left=0.65cm of second] (current) {{\textbf{{{tex_escape(current['title'])}}}\\[3pt]
{tex_escape(current['content'])}}};

\node[protocol, left=0.65cm of current] (future) {{\textbf{{{tex_escape(future['title'])}}}\\[3pt]
{tex_escape(future['content'])}}};

\draw[flow] (initial.east) -- (evaluation.west);
\draw[flow] (evaluation.east) -- (redesign.west);
\draw[flow] (redesign.south) -- (second.north);
\draw[flow] (second.west) -- (current.east);
\draw[prospective] (current.west) -- (future.east);
\end{{tikzpicture}}
\caption{{The documented program transition from 2025 to 2026 and the empirical evaluation plan for outcomes not measured in the present study.}}
\label{{fig:learning-ecosystem}}
\end{{figure*}}
"""


def generate_objectives_table(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\scriptsize", r"\setlength{\tabcolsep}{3pt}",
        r"\caption{Program goals and the participation support condition used in the analysis.}",
        r"\label{tab:educational-objectives-generated}",
        r"\begin{tabularx}{\textwidth}{p{0.04\textwidth} p{0.18\textwidth} p{0.10\textwidth} p{0.18\textwidth} Y Y}",
        r"\toprule", r"\textbf{Code} & \textbf{Goal or condition} & \textbf{Decisions} & \textbf{Position in trajectory} & \textbf{First and second evidence} & \textbf{Inference boundary} \\", r"\midrule",
    ]
    for index, row in enumerate(rows):
        combined = f"First iteration: {row['initial_grounding']} Second iteration: {row['later_operationalization']}"
        values = [row["objective_id"], row["title"], row["decision_ids"], row["trajectory_position"], combined, row["inference_boundary"]]
        lines.append(" & ".join(tex_escape(value) for value in values) + r" \\")
        if index < len(rows) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_decisions_table(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\scriptsize", r"\setlength{\tabcolsep}{3pt}",
        r"\caption{Design decisions mapped to objectives and analytical research questions. Links express intended support rather than causal contribution.}",
        r"\label{tab:design-decisions-generated}",
        r"\begin{tabularx}{\textwidth}{p{0.04\textwidth} p{0.26\textwidth} p{0.13\textwidth} p{0.11\textwidth} p{0.09\textwidth} Y}",
        r"\toprule", r"\textbf{Code} & \textbf{Decision} & \textbf{Lineage} & \textbf{Objectives} & \textbf{RQ} & \textbf{Evidence status and boundary} \\", r"\midrule",
    ]
    for index, row in enumerate(rows):
        lineage = row["lineage"].replace("_", " ")
        evidence_status = row["evidence_status"].replace("_", " ")
        status_boundary = f"{evidence_status}: {row['inference_boundary']}"
        values = [row["decision_id"], row["decision"], lineage, row["objective_codes"], row["rq_links"], status_boundary]
        lines.append(" & ".join(tex_escape(value) for value in values) + r" \\")
        if index < len(rows) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_protocol_table(rows: list[dict[str, str]]) -> str:
    proposition_order = {"P1": 1, "P2": 2, "P3": 3, "P4": 4, "P5": 5}
    ordered_rows = sorted(
        rows,
        key=lambda row: (
            min(proposition_order[code.strip()] for code in row["proposition_ids"].split(",")),
            row["protocol_id"],
        ),
    )
    lines = [
        r"{\scriptsize", r"\setlength{\tabcolsep}{3pt}", r"\renewcommand{\arraystretch}{1.05}",
        r"\begin{longtable}{p{0.045\textwidth} p{0.205\textwidth} p{0.21\textwidth} p{0.265\textwidth} p{0.215\textwidth}}",
        r"\caption{Empirical evaluation plan for P1 through P5 under evaluation principle E1.}\label{tab:future-evaluation} \\",
        r"\toprule", r"\textbf{Code} & \textbf{Unit and primary outcome} & \textbf{Sample and comparator} & \textbf{Measurement and analysis} & \textbf{Inference and missing data} \\", r"\midrule", r"\endfirsthead",
        r"\toprule", r"\textbf{Code} & \textbf{Unit and primary outcome} & \textbf{Sample and comparator} & \textbf{Measurement and analysis} & \textbf{Inference and missing data} \\", r"\midrule", r"\endhead",
        r"\bottomrule", r"\endfoot",
    ]
    for index, row in enumerate(ordered_rows):
        unit_outcome = f"{row['unit_and_timing']} Primary outcome: {row['primary_outcome']}"
        measurement_analysis = f"{row['measurement_and_quality']} Analysis: {row['analysis']}"
        inference = f"{row['decision_rule']} Missing data: {row['missing_data_rule']} Boundary: {row['interpretation_rule']}"
        values = [row["proposition_ids"], unit_outcome, row["sampling_and_comparator"], measurement_analysis, inference]
        lines.append(" & ".join(paper_cell(value) for value in values) + r" \\")
        if index < len(ordered_rows) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\end{longtable}", r"}", ""])
    return "\n".join(lines)


def generate_administrative_table(admin: list[dict[str, str]]) -> str:
    a = administrative_index(admin)
    rows = [
        ("First", "Registration", f"{a['A002']['value_text']} registered people", "First iteration participant registration layer."),
        ("First", "Actual attendance", f"{a['A003']['value_text']} unique attendees", "Final school level attendance; track attendance overlaps."),
        ("First", "Scholarships and reach", f"{a['A004']['value_text']} final allocations; {a['A008']['value_text']} allocated; {a['A005']['value_text']} countries", "Final documented allocation, not payment completion or total program cost."),
        ("Second", "Official form", f"{a['A102']['value_text']} submissions; {a['A103']['value_text']} normalized email keys", "Intake records; submissions and email keys are not interchangeable."),
        ("Second", "Decision dataset", f"{a['A104']['value_text']} applicant identifiers; {a['A106']['value_text']} normalized email keys", "Dataset used for preference and selection analysis."),
        ("Second", "Selection", f"{a['A108']['value_text']} distinct admitted identifiers; {a['A109']['value_text']} track assignments", "Some identifiers were admitted to multiple tracks."),
        ("Second", "Scholarship selection", f"{a['A110']['value_text']} selected identifiers; {a['A111']['value_text']} track assignments; {a['A133']['value_text']} selection stage amount", "Before tax selection stage fields; not acceptance, committed or paid expenditure, or total program cost."),
        ("Second", "International support", f"{a['A112']['value_text']} visa request records; {a['A113']['value_text']} canonical candidate countries", "Process entry and candidate reach layers; not completed support or attendance."),
        ("Second", "Admitted reach", f"{a['A115']['value_text']} canonical admitted countries; {a['A117']['value_text']} countries among scholarship selected identifiers", "Neither is actual attendance."),
        ("Second", "Registration and badges", f"{a['A119']['value_text']} form submissions / {a['A120']['value_text']} email keys; {a['A121']['value_text']} badge rows / {a['A122']['value_text']} names", "Administrative artifacts after selection, not attendance."),
        ("Second", "Organizer reported weekly presence", f"{a['A123']['value_text']} / {a['A124']['value_text']} / {a['A125']['value_text']} headcounts", "Overlapping weekly recollections; no final unique attendance total."),
    ]
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\footnotesize",
        r"\caption{Documented access and participation layers across both editions. Counts retain their units and are not one conversion funnel.}",
        r"\label{tab:administrative-layers-generated}",
        r"\begin{tabularx}{\textwidth}{p{0.07\textwidth} p{0.20\textwidth} p{0.22\textwidth} Y}",
        r"\toprule",
        r"\textbf{Edition} & \textbf{Layer} & \textbf{Count and unit} & \textbf{Interpretation boundary} \\",
        r"\midrule",
    ]
    for index, row in enumerate(rows):
        lines.append(" & ".join(tex_escape(value) for value in row) + r" \\")
        if index in (2,):
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_track_attendance_table(rows: list[dict[str, str]]) -> str:
    by_edition = {
        edition: {row["track"]: row for row in rows if row["edition"] == edition}
        for edition in ("2025", "2026")
    }
    first = by_edition["2025"]
    later = by_edition["2026"]
    table_rows = [
        (
            "2025",
            first["Robotics"]["value_text"],
            first["Artificial Intelligence"]["value_text"],
            first["Digital Twins"]["value_text"],
            r"--",
            str(sum(integer(row, "count") for row in first.values())),
            str(max(integer(row, "count") for row in first.values())),
        ),
        (
            "2026",
            later["Robotics"]["value_text"],
            later["Artificial Intelligence"]["value_text"],
            r"--",
            later["GEO"]["value_text"],
            f"about {sum(integer(row, 'count') for row in later.values())}",
            str(max(integer(row, "count") for row in later.values())),
        ),
    ]
    lines = [
        r"\begin{table*}[!t]",
        r"\centering",
        r"\small",
        r"\caption{Reported weekly presence by track and edition. The 2025 values are published track attendances, whereas the 2026 values are organizer headcounts and the GEO value is approximate. People attending several weeks contribute once to each corresponding track count.}",
        r"\label{tab:track-attendance}",
        r"\begin{tabular}{lrrrrrr}",
        r"\toprule",
        r"\textbf{Edition} & \textbf{\emph{ROB}} & \textbf{\emph{AI}} & \textbf{\emph{DT}} & \textbf{\emph{GEO}} & \textbf{Cumulative weekly presence} & \textbf{Peak week} \\",
        r"\midrule",
    ]
    for row in table_rows:
        lines.append(" & ".join(row) + r" \\")
    lines.extend([r"\bottomrule", r"\end{tabular}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_scale_tikz(rows: list[dict[str, str]]) -> str:
    order = ["SC06", "SC02", "SC05", "SC04", "SC01", "SC03"]
    by_id = {row["indicator_id"]: row for row in rows}
    ordered = [by_id[indicator_id] for indicator_id in order]
    labels = [row["canonical_measure"] for row in ordered]
    symbolic = ",".join(labels)
    coordinates = "\n".join(
        f"    ({row['later_index_first_100']},{row['canonical_measure']})"
        for row in ordered
        if row["indicator_id"] != "SC06"
    )
    speaker = by_id["SC06"]
    annotations = []
    for row in ordered:
        value = float(row["later_index_first_100"])
        change = float(row["percent_change"])
        rounded = int(round(abs(change)))
        sign = "+" if change >= 0 else r"$-$"
        prefix = r"$\approx$" if row["indicator_id"] == "SC04" else ""
        anchor = "west" if change >= 0 else "east"
        x = value + 2 if change >= 0 else value - 2
        annotations.append(
            rf"\node[font=\scriptsize,anchor={anchor}] at (axis cs:{x:.1f},{row['canonical_measure']}) {{{prefix}{sign}{rounded}\%}};"
        )
    annotations_text = "\n".join(annotations)
    return rf"""\begin{{figure*}}[!htbp]
\centering
\begin{{tikzpicture}}
\begin{{axis}}[
    xbar,
    width=0.84\textwidth,
    height=7.0cm,
    xmin=0,
    xmax=460,
    xtick={{0,50,100,150,200,250,300,350,400,450}},
    xlabel={{2026 index, with 2025 set to 100 for each indicator}},
    symbolic y coords={{{symbolic}}},
    ytick={{{symbolic}}},
    y dir=reverse,
    enlarge y limits=0.13,
    xmajorgrids=true,
    grid style={{spgray!18}},
    axis line style={{spgray!65}},
    tick label style={{font=\small}},
    label style={{font=\small}},
    bar width=11pt,
    clip=false
]
\addplot[fill=spblue!72,draw=spblue!95!black,bar shift=0pt] coordinates {{
{coordinates}
}};
\addplot[fill=spgray!48,draw=spgray!85,bar shift=0pt] coordinates {{
    ({speaker['later_index_first_100']},{speaker['canonical_measure']})
}};

\draw[spgray,densely dashed,thick]
    (axis cs:100,{labels[0]}) --
    (axis cs:100,{labels[-1]});
\node[font=\scriptsize,spgray,anchor=south] at (rel axis cs:0.217,1.01) {{2025 baseline}};

{annotations_text}
\end{{axis}}
\end{{tikzpicture}}
\caption{{Relative change in six program scale indicators expressed with common units. Each bar reports the 2026 value as an index, with the corresponding 2025 value set to 100. Official intake individuals are 153 and 444, intake countries are 32 and 69, final scholarship allocations are 54 and 220, cumulative weekly presence is 199 and about 380, peak weekly presence is 86 and 140, and confirmed program speakers are 58 and 55. The later presence values are organizer headcounts, with GEO reported approximately; people present in several weeks contribute to each corresponding weekly count.}}
\label{{fig:scale-trajectory}}
\end{{figure*}}
"""


def track_stage(rows: list[dict[str, str]], edition: str, stage: str) -> list[dict[str, str]]:
    return sorted(
        [row for row in rows if row["edition"] == edition and row["stage"] == stage],
        key=lambda row: integer(row, "track_order"),
    )


def speaker_stage(rows: list[dict[str, str]], edition: str) -> list[dict[str, str]]:
    order = {"Industry-classified": 0, "Other sectors": 1}
    return sorted([row for row in rows if row["edition"] == edition], key=lambda row: order[row["category"]])


def svg_learning_ecosystem(rows: list[dict[str, str]]) -> str:
    def wrap_text(value: str, limit: int) -> list[str]:
        lines: list[str] = []
        current: list[str] = []
        for word in value.split():
            candidate = " ".join(current + [word])
            if current and len(candidate) > limit:
                lines.append(" ".join(current))
                current = [word]
            else:
                current.append(word)
        if current:
            lines.append(" ".join(current))
        return lines

    by_id = {row["stage_id"]: row for row in rows}
    boundary = by_id["inference_boundary"]
    top_ids = ["initial_program", "initial_evidence", "documented_decisions"]
    bottom_ids = ["prospective_protocol", "later_evidence", "later_program"]
    colors = {
        "initial_program": ("#e8f0fb", "#2454a6"),
        "initial_evidence": ("#fff7ed", "#c76e2c"),
        "documented_decisions": ("#e7f6ef", "#2f8f6b"),
        "later_program": ("#e8f0fb", "#2454a6"),
        "later_evidence": ("#fff7ed", "#c76e2c"),
        "prospective_protocol": ("#f4f6f8", "#5c6670"),
    }
    width, height = 1320, 670
    margin, gap, box_w, box_h = 45, 55, 373, 205
    top_y, bottom_y = 35, 285
    parts = [
        f'<svg xmlns="http://www.w3.org/2000/svg" width="{width}" height="{height}" viewBox="0 0 {width} {height}">',
        f'<rect x="0" y="0" width="{width}" height="{height}" fill="#ffffff"/>',
        '<defs><marker id="arrow" markerWidth="10" markerHeight="8" refX="9" refY="4" orient="auto"><path d="M0,0 L10,4 L0,8 z" fill="#52606d"/></marker></defs>',
        '<style>text{font-family:Arial,sans-serif;fill:#1f2933}.heading{font-size:18px;font-weight:700}.model{font-size:16px;font-weight:700}.body{font-size:15px}.flow{stroke:#52606d;stroke-width:3;marker-end:url(#arrow)}.prospective{stroke:#2f8f6b;stroke-width:3;stroke-dasharray:10 8;marker-end:url(#arrow)}</style>',
    ]
    positions: dict[str, tuple[float, float]] = {}
    for row_y, stage_ids in ((top_y, top_ids), (bottom_y, bottom_ids)):
        for index, stage_id in enumerate(stage_ids):
            stage = by_id[stage_id]
            fill, stroke = colors[stage_id]
            x = margin + index * (box_w + gap)
            positions[stage_id] = (x, row_y)
            parts.append(f'<rect x="{x}" y="{row_y}" width="{box_w}" height="{box_h}" rx="4" fill="{fill}" stroke="{stroke}" stroke-width="2"/>')
            parts.append(f'<text class="heading" x="{x + box_w/2}" y="{row_y + 34}" text-anchor="middle">{html.escape(stage["stage_title"])}</text>')
            body_y = row_y + 68
            for line_index, line in enumerate(wrap_text(stage["content"], 44)):
                parts.append(f'<text class="body" x="{x + box_w/2}" y="{body_y + line_index * 22}" text-anchor="middle">{html.escape(line)}</text>')
    for left_id, right_id in zip(top_ids, top_ids[1:]):
        left_x, left_y = positions[left_id]
        right_x, _ = positions[right_id]
        parts.append(f'<line class="flow" x1="{left_x + box_w + 3}" y1="{left_y + box_h/2}" x2="{right_x - 3}" y2="{left_y + box_h/2}"/>')
    for left_id, right_id in zip(bottom_ids, bottom_ids[1:]):
        left_x, left_y = positions[left_id]
        right_x, _ = positions[right_id]
        line_class = "prospective" if left_id == "prospective_protocol" else "flow"
        parts.append(f'<line class="{line_class}" x1="{right_x - 3}" y1="{left_y + box_h/2}" x2="{left_x + box_w + 3}" y2="{left_y + box_h/2}"/>')
    redesign_x, redesign_y = positions["documented_decisions"]
    later_x, later_y = positions["later_program"]
    parts.append(f'<line class="flow" x1="{redesign_x + box_w/2}" y1="{redesign_y + box_h + 3}" x2="{later_x + box_w/2}" y2="{later_y - 3}"/>')
    parts.append('<rect x="90" y="535" width="1140" height="100" rx="4" fill="#f4f6f8" stroke="#5c6670" stroke-width="2"/>')
    parts.append(f'<text class="heading" x="660" y="568" text-anchor="middle">{html.escape(boundary["stage_title"])}</text>')
    for line_index, line in enumerate(wrap_text(boundary["content"], 130)):
        parts.append(f'<text class="body" x="660" y="{598 + line_index * 22}" text-anchor="middle">{html.escape(line)}</text>')
    parts.append('</svg>')
    return "\n".join(parts) + "\n"


def results_summary(
    admin: list[dict[str, str]],
    tracks: list[dict[str, str]],
    binary: list[dict[str, str]],
    likert: list[dict[str, str]],
    slots: list[dict[str, str]],
    speaker_composition: list[dict[str, str]],
) -> str:
    a = administrative_index(admin)
    second_pref = track_stage(tracks, "second", "preferences")
    theory = next(row for row in likert if row["item_id"] == "QL05")
    theory_counts = [integer(theory, f"rating_{rating}") for rating in range(1, 6)]
    speaker = next(row for row in binary if row["indicator_id"] == "LBI01")
    network = next(row for row in binary if row["indicator_id"] == "LBI02")
    initial_speakers = {row["category"]: integer(row, "count") for row in speaker_stage(speaker_composition, "initial")}
    later_speakers = {row["category"]: integer(row, "count") for row in speaker_stage(speaker_composition, "later")}
    pref_lines = "\n".join(
        f"- {row['track']}: {integer(row, 'count')}/737 preferences ({100*integer(row, 'count')/737:.1f}%)."
        for row in second_pref
    )
    return f"""# Generated Results Summary

## Program goals and documented transition

- O1 to O4 are the founding goals retained from the published program model.
- EP1 represents mission oriented practical experience as a priority emerging from first edition evaluation.
- EP2 represents structured face to face dialogue and networking as a priority emerging from first edition evaluation.
- S1 represents support for international participation in person, not an educational outcome.
- D1 to D11 translate goals into program decisions with explicit RQ links and evidence boundaries.
- D6, D7, and D11 have complete process traces connecting a prior need, a later decision, and implementation.
- The evidence documents one transition from 2025 to 2026, not a recurring improvement cycle.
- The seven prospective tests specify evidence required for claims that the current case cannot establish.

## First edition

- 153 registered people; 128 unique attendees; 54 scholarships; EUR 70,950 final documented scholarship allocation; 32 participant countries.
- 77 participant questionnaire responses, corresponding to 60.2% of 128 actual attendees.

## Second edition administrative layers

- {a['A102']['value_text']} official form submissions represented by {a['A103']['value_text']} normalized email keys.
- {a['A104']['value_text']} applicant IDs in the decision dataset and {a['A106']['value_text']} normalized email keys in that source.
- {a['A108']['value_text']} distinct admitted IDs and {a['A110']['value_text']} distinct IDs selected for scholarship allocation.
- {a['A133']['value_text']} is the pre-tax selection-stage field total; it is not proof of acceptance, commitment, payment, or total program cost.
- {a['A112']['value_text']} visa-request records.
- {a['A113']['value_text']} canonical candidate countries, {a['A115']['value_text']} admitted countries, and {a['A117']['value_text']} countries among scholarship-selected IDs.
- Organizer-reported weekly physical headcounts: 140, 140, and about 100. These are overlapping weekly values, not unique attendance.

## Track preferences

{pref_lines}

Preferences were collected after the portfolio was announced and may include multiple tracks per applicant.

## Speaker-record composition

- First iteration: {initial_speakers['Industry-classified']} industry-classified and {initial_speakers['Other sectors']} other-sector contributing speakers (58 total).
- Second iteration: {later_speakers['Industry-classified']} industry-classified and {later_speakers['Other sectors']} other-sector accepted-speaker records (55 total).
- The stages and source classifications differ; these values audit source bounded composition, not delivered exposure, finer sector representation, or sector effects.

## Questionnaire-row indicators

- Second-edition speaker interaction: {format_indicator(integer(speaker, 'successes'), integer(speaker, 'n'))}.
- Second-edition professional-network expansion: {format_indicator(integer(network, 'successes'), integer(network, 'n'))}.
- Theory-practice balance numeric distribution across categories 1--5: {', '.join(str(value) for value in theory_counts)}. Endpoint labels were verified against the retained form definition.

The package retains complete category counts for each edition and a harmonized descriptive summary of 13 common closed items. The summary does not support causal, population, or improvement inference.

## Practical formats

- {len(slots)} explicit timetable slots representing {len({row['format_id'] for row in slots})} titled formats: six Robotics slots, one Artificial Intelligence slot, and four GEO slots.
- Four Robotics rows are repeated slots of one Rover Systems Lab format.
- These counts establish scheduled provision only.
"""


def validation_report(audit: Audit) -> str:
    passed = sum(ok for _, ok, _ in audit.items)
    lines = [
        "# Validation Report",
        "",
        f"Overall status: **{'PASS' if audit.passed else 'FAIL'}**",
        "",
        f"Checks passed: {passed}/{len(audit.items)}",
        "",
        "PASS means that the released tables satisfy the declared arithmetic, schema, provenance-label, and internal-consistency checks. It does not independently validate author coding against source artifacts that are not distributed.",
        "",
        "| Status | Check | Detail |",
        "|---|---|---|",
    ]
    for name, ok, detail in audit.items:
        safe_detail = detail.replace("|", "\\|")
        lines.append(f"| {'PASS' if ok else 'FAIL'} | {name} | {safe_detail} |")
    lines.append("")
    lines.append("The build exits non-zero when any row is marked FAIL.")
    lines.append("")
    return "\n".join(lines)


def validation_payload(audit: Audit) -> dict[str, object]:
    passed = sum(ok for _, ok, _ in audit.items)
    return {
        "overall_status": "PASS" if audit.passed else "FAIL",
        "checks_passed": passed,
        "checks_total": len(audit.items),
        "scope": "Arithmetic, schema, provenance label, privacy, and internal consistency checks over the released artifacts.",
        "boundary": "A PASS does not independently validate author coding against restricted source artifacts.",
        "checks": [
            {"status": "PASS" if ok else "FAIL", "check": name, "detail": detail}
            for name, ok, detail in audit.items
        ],
    }


def results_summary_payload(
    admin: list[dict[str, str]],
    tracks: list[dict[str, str]],
    slots: list[dict[str, str]],
    questionnaire_common: list[dict[str, str]],
) -> dict[str, object]:
    a = administrative_index(admin)
    return {
        "editions": 2,
        "first_edition": {
            "registered_people": integer(a["A002"], "value_numeric"),
            "actual_attendees": integer(a["A003"], "value_numeric"),
            "scholarships": integer(a["A004"], "value_numeric"),
            "participant_countries": integer(a["A005"], "value_numeric"),
            "returned_participant_forms": integer(a["A006"], "value_numeric"),
        },
        "second_edition": {
            "formal_submissions": integer(a["A102"], "value_numeric"),
            "decision_dataset_applicant_ids": integer(a["A104"], "value_numeric"),
            "admitted_applicant_ids": integer(a["A108"], "value_numeric"),
            "scholarship_selected_ids": integer(a["A110"], "value_numeric"),
            "visa_request_records": integer(a["A112"], "value_numeric"),
            "candidate_countries_in_decision_dataset": integer(a["A113"], "value_numeric"),
            "returned_participant_forms": integer(a["A130"], "value_numeric"),
        },
        "track_preferences": {
            row["track"]: integer(row, "count")
            for row in tracks
            if row["edition"] == "second" and row["stage"] == "preferences"
        },
        "practical_provision": {
            "timetable_slots": len(slots),
            "named_formats": len({row["format_id"] for row in slots}),
            "scheduled_hours": sum(
                (int(row["end_time"].split(":")[0]) * 60 + int(row["end_time"].split(":")[1]))
                - (int(row["start_time"].split(":")[0]) * 60 + int(row["start_time"].split(":")[1]))
                for row in slots
            ) / 60,
        },
        "harmonized_questionnaire_items": len(questionnaire_common),
        "interpretation_boundary": "All questionnaire values describe separate returned form sets and all administrative values retain their original stage and unit.",
    }


def generate_access_process_table(rows: list[dict[str, str]]) -> str:
    by_id = {row["metric_id"]: row for row in rows}
    selected = [
        ("Decision records", "AP01"),
        ("Admitted", "AP02"),
        ("Scholarship requests", "AP06"),
        ("Scholarship selections", "AP07"),
        ("Visa requests among admitted", "AP09"),
        ("Exact admission to registration links", "AP15"),
        ("Admission records without an exact registration link", "AP16"),
    ]
    lines = [
        r"\begin{table}[!t]", r"\centering", r"\scriptsize",
        r"\caption{Later access process aggregates. Stages retain their source units and do not form a confirmed attendance funnel.}",
        r"\label{tab:access-process-generated}",
        r"\begin{tabularx}{\columnwidth}{Y r p{0.30\columnwidth}}",
        r"\toprule", r"\textbf{Process layer} & \textbf{Count} & \textbf{Unit} " + r"\\", r"\midrule",
    ]
    for index, (label, metric_id) in enumerate(selected):
        row = by_id[metric_id]
        lines.append(f"{tex_escape(label)} & {row['value']} & {tex_escape(row['unit'])} " + r"\\")
        if index != len(selected) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table}", ""])
    return "\n".join(lines)


def generate_capacity_pressure_table(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table}[!t]", r"\centering", r"\scriptsize",
        r"\caption{Later admission assignment pressure by track. The operating limit is organizer reported; assignments are not attendance.}",
        r"\label{tab:capacity-pressure-generated}",
        r"\begin{tabular}{lrrr}",
        r"\toprule", r"\textbf{Track} & \textbf{Admitted} & \textbf{Eligible} & \textbf{Limit} " + r"\\", r"\midrule",
    ]
    for row in rows:
        lines.append(
            f"{tex_escape(row['track'])} & {row['admitted_assignments']} & "
            f"{row['eligible_not_admitted_assignments']} & {row['documented_operating_limit']} " + r"\\"
        )
    lines.extend([r"\bottomrule", r"\end{tabular}", r"\end{table}", ""])
    return "\n".join(lines)


def generate_scholarship_model_table(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\scriptsize",
        r"\caption{Traceability between the published scholarship criteria and the operational scoring workbook.}",
        r"\label{tab:scholarship-model-generated}",
        r"\begin{tabularx}{\textwidth}{p{0.13\textwidth} p{0.24\textwidth} p{0.09\textwidth} p{0.18\textwidth} Y}",
        r"\toprule", r"\textbf{Group} & \textbf{Criterion} & \textbf{Maximum} & \textbf{Workbook field} & \textbf{Boundary} " + r"\\", r"\midrule",
    ]
    for index, row in enumerate(rows):
        maximum = row["maximum_points"] or "Not weighted"
        values = [row["applicant_group"], row["criterion"], maximum, row["operational_workbook_field"], row["boundary"]]
        lines.append(" & ".join(tex_escape(value) for value in values) + r" \\")
        if index != len(rows) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_speaker_track_table(rows: list[dict[str, str]]) -> str:
    by_key = {(row["track"], row["sector_classification"]): row["accepted_records"] for row in rows}
    lines = [
        r"\begin{table}[!t]", r"\centering", r"\scriptsize",
        r"\caption{Later accepted speaker records by track and source classification. These are not verified delivered speakers.}",
        r"\label{tab:speaker-track-generated}",
        r"\begin{tabular}{lrrr}",
        r"\toprule", r"\textbf{Track} & \textbf{Industry} & \textbf{Other} & \textbf{Total} " + r"\\", r"\midrule",
    ]
    for track in ("Robotics", "AI", "GEO"):
        industry = int(by_key[(track, "Industry classified")])
        other = int(by_key[(track, "Other sectors")])
        lines.append(f"{track} & {industry} & {other} & {industry + other} " + r"\\")
    lines.extend([r"\bottomrule", r"\end{tabular}", r"\end{table}", ""])
    return "\n".join(lines)


def generate_program_objectives_paper(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\scriptsize", r"\setlength{\tabcolsep}{3pt}",
        r"\renewcommand{\arraystretch}{1.04}",
        r"\caption{Founding objectives, emergent design priorities, and the participation support condition.}",
        r"\label{tab:program-objectives}",
        r"\begin{tabularx}{\textwidth}{p{0.04\textwidth} p{0.19\textwidth} p{0.18\textwidth} p{0.31\textwidth} Y}",
        r"\toprule", r"\textbf{Code} & \textbf{Status and first documentation} & \textbf{Program element} & \textbf{Principal decisions} & \textbf{Evidence available} \\", r"\midrule",
    ]
    for index, row in enumerate(rows):
        status_and_documentation = f"{row['analytical_status']}. {row['first_documented']}"
        values = [row["objective_id"], status_and_documentation, row["program_goal_or_condition"], row["principal_decisions"], row["evidence_available"]]
        lines.append(" & ".join(paper_cell(value) for value in values) + r" \\")
        if index < len(rows) - 1:
            lines.append(r"\midrule" if row["objective_id"] in {"O4", "EP2"} else r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_data_corpus_paper(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\scriptsize", r"\setlength{\tabcolsep}{3pt}",
        r"\renewcommand{\arraystretch}{1.05}",
        r"\caption{Data sources, units, and analytical roles.}", r"\label{tab:data-corpus}",
        r"\begin{tabularx}{\textwidth}{p{0.07\textwidth} p{0.22\textwidth} p{0.27\textwidth} p{0.08\textwidth} Y}",
        r"\toprule", r"\textbf{Edition} & \textbf{Source} & \textbf{Unit and coverage} & \textbf{RQ} & \textbf{Analytical role} \\", r"\midrule",
    ]
    for index, row in enumerate(rows):
        values = [row["edition"], row["source"], row["unit_and_coverage"], row["rq"], row["analytical_role"]]
        lines.append(" & ".join(paper_cell(value) for value in values) + r" \\")
        if index < len(rows) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_curricular_connections_paper(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\scriptsize", r"\setlength{\tabcolsep}{3pt}",
        r"\renewcommand{\arraystretch}{1.05}",
        r"\caption{Documentary mapping of curricular connections in the published 2025 program and official 2026 agenda.}",
        r"\label{tab:space-integration}",
        r"\begin{tabularx}{\textwidth}{p{0.07\textwidth} p{0.12\textwidth} p{0.25\textwidth} p{0.36\textwidth} Y}",
        r"\toprule", r"\textbf{Edition} & \textbf{Program area} & \textbf{Documented topics and perspectives} & \textbf{Examples in program records} & \textbf{Record status} \\", r"\midrule",
    ]
    for index, row in enumerate(rows):
        area_codes = {"Robotics": "ROB", "Digital Twins": "DT", "AI": "AI", "GEO": "GEO"}
        values = [paper_cell(row["edition"]), rf"\emph{{{area_codes[row['program_area']]}}}", paper_cell(row["perspectives_present"]), paper_cell(row["examples_in_records"]), paper_cell(row["record_status"])]
        lines.append(" & ".join(values) + r" \\")
        if index < len(rows) - 1:
            lines.append(r"\midrule" if rows[index + 1]["edition"] != row["edition"] else r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_participation_support_paper(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\scriptsize", r"\setlength{\tabcolsep}{3pt}",
        r"\caption{Participation scale and support functions across the two editions.}",
        r"\label{tab:administrative-layers}",
        r"\begin{tabularx}{\textwidth}{p{0.07\textwidth} p{0.20\textwidth} p{0.35\textwidth} Y}",
        r"\toprule", r"\textbf{Edition} & \textbf{Stage} & \textbf{Recorded scale} & \textbf{Organizational function} \\", r"\midrule",
    ]
    for index, row in enumerate(rows):
        values = [row["edition"], row["stage"], row["recorded_scale"], row["organizational_function"]]
        lines.append(" & ".join(paper_cell(value) for value in values) + r" \\")
        if index < len(rows) - 1:
            if rows[index + 1]["edition"] != row["edition"]:
                lines.append(r"\midrule")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_rover_systems_lab_paper(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[t]", r"\centering", r"\small",
        r"\caption{Documented and reported implementation, intended educational purposes, and prospective reuse of the \emph{SPACERAISE Academy Rover Systems Lab}.}",
        r"\label{tab:rover-systems-lab}",
        r"\begin{tabularx}{\textwidth}{p{0.15\textwidth} p{0.27\textwidth} p{0.25\textwidth} Y}",
        r"\toprule", r"\textbf{Design element} & \textbf{Documented or reported 2026 implementation} & \textbf{Intended educational purpose} & \textbf{Prospective reuse} " + r"\\", r"\midrule",
    ]
    for index, row in enumerate(rows):
        values = [row["design_element"], row["documented_2026_realization"], row["educational_focus"], row["reuse_through_academy"]]
        lines.append(" & ".join(paper_cell(value) for value in values) + r" \\")
        if index < len(rows) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_explanatory_synthesis_paper(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\scriptsize", r"\setlength{\tabcolsep}{3pt}",
        r"\renewcommand{\arraystretch}{1.05}",
        r"\caption{Author synthesis of cross edition case observations into provisional design lessons.}", r"\label{tab:explanatory-synthesis}",
        r"\begin{tabularx}{\textwidth}{p{0.045\textwidth} p{0.225\textwidth} p{0.22\textwidth} p{0.34\textwidth} Y}",
        r"\toprule", r"\textbf{Code} & \textbf{Case observations} & \textbf{Interpreted design tension} & \textbf{Provisional design lesson} & \textbf{Claim scope and formalization} \\", r"\midrule",
    ]
    for index, row in enumerate(rows):
        values = [
            paper_cell(row["lesson_id"]),
            paper_track_terms(row["evidence_across_editions"]),
            paper_cell(row["design_tension"]),
            paper_cell(row["design_lesson"]),
            paper_cell(row["scope_and_formalization"]),
        ]
        lines.append(" & ".join(values) + r" \\")
        if index < len(rows) - 1:
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table*}", ""])
    return "\n".join(lines)


def generate_questionnaire_summary_paper(rows: list[dict[str, str]]) -> str:
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\scriptsize", r"\setlength{\tabcolsep}{4pt}",
        r"\caption{Descriptive summary of the 13 harmonized closed questionnaire items. Values are returned forms satisfying the declared rule.}",
        r"\label{tab:returned-questionnaires}",
        r"\begin{tabularx}{\textwidth}{p{0.28\textwidth} p{0.20\textwidth} r r Y}",
        r"\toprule", r"\textbf{Program aspect} & \textbf{Response rule} & \textbf{2025, $n=77$} & \textbf{2026, $n=82$} & \textbf{Analytical use} \\", r"\midrule",
    ]
    for row in rows:
        first = f"{row['first_successes']} ({row['first_percent']}\\%)"
        later = f"{row['later_successes']} ({row['later_percent']}\\%)"
        rule = "Categories 4 or 5" if row["response_rule"] == "upper categories 4 or 5" else row["response_rule"]
        values = [paper_cell(row["construct"]), paper_cell(rule), first, later, paper_cell(row["analytical_use"])]
        lines.append(" & ".join(values) + r" \\")
    lines.extend([
        r"\bottomrule", r"\end{tabularx}", r"\vspace{2pt}",
        r"\parbox{0.97\textwidth}{\scriptsize Numeric items use categories 4 or 5; categorical items use the exact Yes category. Question wording and response definitions were checked against both form definitions. For theory and practice, the later form used ``appropriate'' rather than ``balanced'' in its endpoint labels. The two columns describe different returned form sets and do not establish improvement, population change, or a program effect.}",
        r"\end{table*}", "",
    ])
    return "\n".join(lines)


def generate_participant_profiles_paper(rows: list[dict[str, str]]) -> str:
    selected = [
        row for row in rows
        if (row["edition"], row["stage"]) in {("first", "actual attendees"), ("second", "admitted IDs")}
    ]
    groups: dict[tuple[str, str], list[dict[str, str]]] = defaultdict(list)
    for row in selected:
        groups[(row["edition"], row["stage"])].append(row)
    lines = [
        r"\begin{table}[H]", r"\centering", r"\scriptsize",
        r"\caption{Learner profiles in 2025 attendance and 2026 admissions.}", r"\label{tab:participant-profiles}",
        r"\begin{tabularx}{\columnwidth}{p{0.12\columnwidth} p{0.20\columnwidth} r Y}",
        r"\toprule", r"\textbf{Edition} & \textbf{Stage} & \textbf{Total} & \textbf{Profile counts} \\", r"\midrule",
    ]
    profile_order = {"Master's students": 0, "PhD students": 1, "Researchers": 2, "Practitioners": 3}
    for edition, stage in (("first", "actual attendees"), ("second", "admitted IDs")):
        group = sorted(groups[(edition, stage)], key=lambda row: profile_order[row["profile"]])
        total = group[0]["denominator"]
        labels = {"Master's students": "master's students", "PhD students": "doctoral students", "Researchers": "researchers", "Practitioners": "practitioners"}
        counts = ", ".join(f"{row['count']} {labels[row['profile']]}" for row in group)
        values = ["2025" if edition == "first" else "2026", "Actual attendance" if edition == "first" else "Admission", total, counts]
        lines.append(" & ".join(tex_escape(value) for value in values) + r" \\")
        if edition == "first":
            lines.append(r"\addlinespace")
    lines.extend([r"\bottomrule", r"\end{tabularx}", r"\end{table}", ""])
    return "\n".join(lines)


def generate_track_capacity_paper(tracks: list[dict[str, str]], capacity: list[dict[str, str]]) -> str:
    preferences = {
        ("AI" if row["track"] == "Artificial Intelligence" else row["track"]): row["count"]
        for row in tracks
        if row["edition"] == "second" and row["stage"] == "preferences"
    }
    by_track = {row["track"]: row for row in capacity}
    lines = [
        r"\begin{table}[H]", r"\centering", r"\scriptsize",
        r"\caption{Track preferences and selection assignments in 2026.}", r"\label{tab:track-capacity}",
        r"\begin{tabular}{lrrr}", r"\toprule", r"\textbf{Track} & \textbf{Preferences} & \textbf{Admitted} & \textbf{Eligible, not admitted} \\", r"\midrule",
    ]
    for track in ("Robotics", "AI", "GEO"):
        row = by_track[track]
        track_code = {"Robotics": "ROB", "AI": "AI", "GEO": "GEO"}[track]
        lines.append(rf"\emph{{{track_code}}} & {preferences[track]} & {row['admitted_assignments']} & {row['eligible_not_admitted_assignments']} " + r"\\")
    lines.extend([
        r"\bottomrule", r"\end{tabular}", r"\vspace{2pt}",
        r"\parbox{0.95\columnwidth}{\scriptsize Applicants could select several tracks, but each selected track was evaluated independently. Admission therefore applied only to the track or tracks for which the applicant was selected.}",
        r"\end{table}", "",
    ])
    return "\n".join(lines)


def generate_track_sequence_demand_paper(rows: list[dict[str, str]]) -> str:
    selected_by_edition = {
        "2025": sorted(
            (row for row in rows if row["edition"] == "2025" and row["source_stage"] == "entry registrations"),
            key=lambda row: integer(row, "path_order"),
        ),
        "2026": sorted(
            (row for row in rows if row["edition"] == "2026" and row["source_stage"] == "applicant preferences"),
            key=lambda row: integer(row, "path_order"),
        ),
    }
    class_labels = {
        "single track": "Single track",
        "adjacent tracks": "Adjacent tracks",
        "non adjacent tracks": "Non adjacent tracks",
        "complete sequence": "Complete sequence",
    }
    lines = [
        r"\begin{table*}[!t]", r"\centering", r"\scriptsize",
        r"\caption{Exact entry track paths by edition.}",
        r"\label{tab:track-sequence-demand}",
        r"\begin{tabular}{cllrr}",
        r"\toprule", r"\textbf{Edition} & \textbf{Declared path} & \textbf{Position} & \textbf{Count} & \textbf{Share} \\", r"\midrule",
    ]
    for edition, selected in selected_by_edition.items():
        denominator = integer(selected[0], "denominator")
        for row in selected:
            path = " + ".join(rf"\emph{{{code}}}" for code in row["path_code"].split("_"))
            share = pct(integer(row, "count") / denominator)
            lines.append(f"{edition} & {path} & {class_labels[row['sequence_class']]} & {row['count']} & {share}\\% " + r"\\")
        if edition == "2025":
            lines.append(r"\addlinespace")
    lines.extend([
        r"\bottomrule", r"\end{tabular}", r"\vspace{2pt}",
        r"\parbox{0.96\textwidth}{\scriptsize Patterns are mutually exclusive. The 2025 paths are reconstructed from published aggregates for 153 registered people: 62 of 66 people selecting several tracks included \emph{AI}. The 2026 paths are derived from 444 unique applicant identifiers: 194 of 197 applicants selecting several tracks included \emph{AI}. Entry choices describe declared demand, not attendance or learning.}",
        r"\end{table*}", "",
    ])
    return "\n".join(lines)


def generate_proposition_map_table(rows: list[dict[str, str]]) -> str:
    lines = [
        r"{\scriptsize", r"\setlength{\tabcolsep}{3pt}", r"\renewcommand{\arraystretch}{1.05}",
        r"\begin{longtable}{p{0.045\textwidth} p{0.23\textwidth} p{0.245\textwidth} p{0.22\textwidth} p{0.20\textwidth}}",
        r"\caption{Case motivated provisional propositions, observed basis, and unmeasured outcomes.}\label{tab:proposition-evidence-generated} \\",
        r"\toprule", r"\textbf{Code} & \textbf{Candidate relationship} & \textbf{Observed case basis} & \textbf{Outcome not measured} & \textbf{Boundary and required test} \\", r"\midrule", r"\endfirsthead",
        r"\toprule", r"\textbf{Code} & \textbf{Candidate relationship} & \textbf{Observed case basis} & \textbf{Outcome not measured} & \textbf{Boundary and required test} \\", r"\midrule", r"\endhead",
        r"\bottomrule", r"\endfoot",
    ]
    for row in rows:
        boundary_test = f"{row['claim_boundary']} Test: {row['prospective_protocol_ids']}"
        values = [row["proposition_id"], row["proposition"], row["case_relationship"], row["outcome_not_measured"], boundary_test]
        lines.append(" & ".join(tex_escape(value) for value in values) + r" \\")
        lines.append(r"\addlinespace")
    lines.extend([r"\end{longtable}", r"}", ""])
    return "\n".join(lines)


def svg_program_transition(nodes: list[dict[str, str]], edges: list[dict[str, str]]) -> str:
    by_id = {row["node_id"]: row for row in nodes}
    positions = {
        "initial": (30, 35), "evaluation": (450, 35), "redesign": (870, 35),
        "future": (30, 290), "current": (450, 290), "second": (870, 290),
    }
    colors = {"program": ("#e8f0fb", "#2454a6"), "evidence": ("#fff7ed", "#c76e2c"), "decision": ("#e7f6ef", "#2f8f6b"), "protocol": ("#f4f6f8", "#5c6670")}
    width, height, box_w, box_h = 1260, 535, 360, 190

    def wrapped(value: str, limit: int = 42) -> list[str]:
        lines: list[str] = []
        current: list[str] = []
        for word in value.split():
            if current and len(" ".join(current + [word])) > limit:
                lines.append(" ".join(current))
                current = [word]
            else:
                current.append(word)
        if current:
            lines.append(" ".join(current))
        return lines

    parts = [
        f'<svg xmlns="http://www.w3.org/2000/svg" width="{width}" height="{height}" viewBox="0 0 {width} {height}">',
        '<rect width="100%" height="100%" fill="#ffffff"/>',
        '<defs><marker id="arrow" markerWidth="10" markerHeight="8" refX="9" refY="4" orient="auto"><path d="M0,0 L10,4 L0,8 z" fill="#52606d"/></marker></defs>',
        '<style>text{font-family:Arial,sans-serif;fill:#1f2933}.heading{font-size:18px;font-weight:700}.body{font-size:15px}.edge{stroke:#52606d;stroke-width:3;fill:none;marker-end:url(#arrow)}.prospective{stroke:#2f8f6b;stroke-dasharray:10 8}</style>',
    ]
    for node_id, (x, y) in positions.items():
        node = by_id[node_id]
        fill, stroke = colors[node["node_type"]]
        parts.append(f'<rect x="{x}" y="{y}" width="{box_w}" height="{box_h}" rx="4" fill="{fill}" stroke="{stroke}" stroke-width="2"/>')
        parts.append(f'<text class="heading" x="{x + box_w / 2}" y="{y + 32}" text-anchor="middle">{html.escape(node["title"])}</text>')
        for index, line in enumerate(wrapped(node["content"])):
            parts.append(f'<text class="body" x="{x + box_w / 2}" y="{y + 66 + index * 22}" text-anchor="middle">{html.escape(line)}</text>')
    for edge in edges:
        source, target = edge["source_node"], edge["target_node"]
        sx, sy = positions[source]
        tx, ty = positions[target]
        css = "edge prospective" if edge["edge_type"] == "prospective" else "edge"
        if source == "redesign" and target == "second":
            x1, y1, x2, y2 = sx + box_w / 2, sy + box_h, tx + box_w / 2, ty
        elif sy == ty and sx < tx:
            x1, y1, x2, y2 = sx + box_w, sy + box_h / 2, tx, ty + box_h / 2
        else:
            x1, y1, x2, y2 = sx, sy + box_h / 2, tx + box_w, ty + box_h / 2
        parts.append(f'<line class="{css}" x1="{x1}" y1="{y1}" x2="{x2}" y2="{y2}"/>')
    parts.append('</svg>')
    return "\n".join(parts) + "\n"


def generate_questionnaire_tikz(
    questionnaire_2025: list[dict[str, str]],
    questionnaire_2026: list[dict[str, str]],
) -> str:
    def distribution(rows: list[dict[str, str]]) -> tuple[list[int], int]:
        selected = sorted(
            [row for row in rows if row["item_id"] == "Q07"],
            key=lambda row: int(row["category"]),
        )
        return [integer(row, "count") for row in selected], integer(selected[0], "n")

    first_counts, first_n = distribution(questionnaire_2025)
    later_counts, later_n = distribution(questionnaire_2026)
    first_pct = [f"{100 * count / first_n:.2f}" for count in first_counts]
    later_pct = [f"{100 * count / later_n:.2f}" for count in later_counts]
    plots = [
        ("spred!82", "spred!95!black"),
        ("sporange!78", "sporange!90!black"),
        ("spgray!40", "spgray!75"),
        ("spblue!68", "spblue!90!black"),
        ("spgreen!72", "spgreen!85!black"),
    ]

    lines = [
        r"\begin{figure*}[!t]",
        r"\centering",
        r"\begin{tikzpicture}",
        r"\begin{axis}[",
        r"    xbar stacked,",
        r"    width=0.82\textwidth,",
        r"    height=4.2cm,",
        r"    xmin=0,",
        r"    xmax=100,",
        r"    xlabel={Share of returned forms (\%)},",
        r"    symbolic y coords={2025,2026},",
        r"    y dir=reverse,",
        r"    ytick=data,",
        r"    yticklabel style={font=\small},",
        r"    xtick={0,20,40,60,80,100},",
        r"    xmajorgrids=true,",
        r"    grid style={draw=spgray!18},",
        r"    axis line style={draw=spgray!55},",
        r"    tick style={draw=spgray!55},",
        r"    legend style={at={(0.5,1.08)},anchor=south,legend columns=5,draw=none,font=\scriptsize},",
        r"    legend image code/.code={\draw[#1] (0cm,-0.08cm) rectangle (0.24cm,0.08cm);},",
        r"    bar width=15pt,",
        r"    enlarge y limits=0.42,",
        r"]",
    ]
    for index, (fill, draw) in enumerate(plots):
        lines.append(
            rf"\addplot[fill={fill},draw={draw}] coordinates "
            rf"{{({first_pct[index]},2025) ({later_pct[index]},2026)}};"
        )
    lines.extend([
        r"\legend{Rating 1,Rating 2,Rating 3,Rating 4,Rating 5}",
        r"\end{axis}",
        r"\end{tikzpicture}",
        rf"\caption{{Complete response distributions for the theory and practice item. Counts for ratings 1 through 5 are {', '.join(map(str, first_counts[:-1]))}, and {first_counts[-1]} in 2025 ($n={first_n}$), and {', '.join(map(str, later_counts[:-1]))}, and {later_counts[-1]} in 2026 ($n={later_n}$). The 2025 endpoints used ``balanced'', whereas the 2026 endpoints used ``appropriate''. The distributions describe separate returned form sets and do not estimate change or a redesign effect.}}",
        r"\label{fig:questionnaire-summary}",
        r"\end{figure*}",
        "",
    ])
    return "\n".join(lines)


def svg_theory_practice(questionnaire_2025: list[dict[str, str]], questionnaire_2026: list[dict[str, str]]) -> str:
    colors = ["#b3363d", "#d9822b", "#a7b0b7", "#3975b7", "#2f8f6b"]
    groups = []
    for edition, rows in (("2025", questionnaire_2025), ("2026", questionnaire_2026)):
        selected = sorted([row for row in rows if row["item_id"] == "Q07"], key=lambda row: int(row["category"]))
        groups.append((edition, [integer(row, "count") for row in selected], integer(selected[0], "n")))
    width, height = 1120, 310
    left, bar_w, bar_h = 110, 900, 54
    parts = [
        f'<svg xmlns="http://www.w3.org/2000/svg" width="{width}" height="{height}" viewBox="0 0 {width} {height}">',
        '<rect width="100%" height="100%" fill="#ffffff"/>',
        '<style>text{font-family:Arial,sans-serif;fill:#1f2933}.label{font-size:18px;font-weight:700}.tick{font-size:13px}.legend{font-size:14px}</style>',
    ]
    for group_index, (edition, counts, n) in enumerate(groups):
        y = 95 + group_index * 95
        parts.append(f'<text class="label" x="25" y="{y + 34}">{edition}</text>')
        x = left
        for rating, count in enumerate(counts, start=1):
            segment = bar_w * count / n
            parts.append(f'<rect x="{x:.2f}" y="{y}" width="{segment:.2f}" height="{bar_h}" fill="{colors[rating - 1]}"/>')
            if segment > 42:
                parts.append(f'<text class="tick" x="{x + segment / 2:.2f}" y="{y + 34}" text-anchor="middle" fill="#ffffff">{100 * count / n:.2f}</text>')
            x += segment
    legend_x = 175
    for rating, color in enumerate(colors, start=1):
        x = legend_x + (rating - 1) * 165
        parts.append(f'<rect x="{x}" y="24" width="20" height="14" fill="{color}"/>')
        parts.append(f'<text class="legend" x="{x + 28}" y="37">Rating {rating}</text>')
    parts.append('</svg>')
    return "\n".join(parts) + "\n"


def main() -> int:
    TABLES.mkdir(parents=True, exist_ok=True)
    FIGURES.mkdir(parents=True, exist_ok=True)

    admin = read_csv(DATA / "administrative-layers.csv")
    tracks = read_csv(DATA / "track-selection.csv")
    track_combinations = read_csv(DATA / "track-combination-patterns.csv")
    preference_admission = read_csv(DATA / "preference-admission-multiplicity.csv")
    scale = read_csv(DATA / "program-scale-comparison.csv")
    track_attendance = read_csv(DATA / "track-attendance-comparison.csv")
    binary = read_csv(DATA / "questionnaire-binary-indicators.csv")
    likert = read_csv(DATA / "questionnaire-likert-distributions.csv")
    categorical = read_csv(DATA / "questionnaire-categorical-distributions.csv")
    first_workbook_check = read_csv(DATA / "first-edition-public-workbook-check.csv")
    attendance_patterns = read_csv(DATA / "questionnaire-attendance-patterns.csv")
    country_summary = read_csv(DATA / "country-normalization-summary.csv")
    profiles = read_csv(DATA / "participant-profiles.csv")
    demographics = read_csv(DATA / "registration-demographics.csv")
    agenda = read_csv(DATA / "agenda-classification-summary.csv")
    slots = read_csv(DATA / "practical-slots.csv")
    agenda_coding = read_csv(DATA / "agenda-coding-assertions.csv")
    learning_content = read_csv(DATA / "learning-ecosystem-content.csv")
    speaker_composition = read_csv(DATA / "speaker-composition.csv")
    evidence = read_csv(DESIGN / "evidence-by-rq.csv")
    objectives = read_csv(DESIGN / "educational-objectives.csv")
    decisions = read_csv(DESIGN / "design-decisions.csv")
    protocol = read_csv(DESIGN / "prospective-evaluation-protocol.csv")
    traceability = read_csv(DESIGN / "redesign-traceability.csv")
    comparators = read_csv(DESIGN / "longitudinal-comparators.csv")
    rq1_themes = read_csv(DESIGN / "rq1-theme-frame.csv")
    data_dictionary = read_csv(DATA / "data-dictionary.csv")
    questionnaire_2025 = read_csv(DATA / "questionnaire-distributions-2025.csv")
    questionnaire_2026 = read_csv(DATA / "questionnaire-distributions-2026.csv")
    questionnaire_common = read_csv(DATA / "questionnaire-common-summary.csv")
    open_comment_inventory = read_csv(DATA / "open-comment-inventory.csv")
    access_process = read_csv(DATA / "access-process-aggregates.csv")
    capacity_pressure = read_csv(DATA / "track-capacity-pressure.csv")
    scholarship_model = read_csv(DATA / "scholarship-selection-model.csv")
    scholarship_rule = read_csv(DATA / "scholarship-allocation-rule.csv")
    speaker_by_track = read_csv(DATA / "speaker-composition-by-track.csv")
    propositions = read_csv(DESIGN / "proposition-evidence-map.csv")
    output_traceability = read_csv(DESIGN / "paper-output-traceability.csv")
    explanatory_synthesis = read_csv(DESIGN / "explanatory-synthesis.csv")
    curricular_connections = read_csv(DATA / "curricular-connections.csv")
    corpus_metadata = read_csv(DATA / "corpus-metadata.csv")
    speaker_detailed = read_csv(DATA / "speaker-composition-detailed.csv")
    networking_design = read_csv(DATA / "networking-activity-design.csv")
    communication_use = read_csv(DATA / "communication-and-networking.csv")
    rover_design = read_csv(DATA / "rover-systems-lab-design.csv")
    expertise_fields = read_csv(DATA / "application-expertise-fields.csv")
    source_schemas = read_csv(GOVERNANCE / "source-schema-manifest.csv")
    transition_nodes = read_csv(DESIGN / "program-transition-nodes.csv")
    transition_edges = read_csv(DESIGN / "program-transition-edges.csv")
    paper_objectives = read_csv(DESIGN / "program-objectives-paper.csv")
    paper_corpus = read_csv(DESIGN / "data-corpus-paper.csv")
    paper_participation = read_csv(DATA / "participation-support-paper.csv")
    rq_document = "\n".join(row["question"] for row in evidence)

    audit = validate_data(
        admin, tracks, track_combinations, preference_admission, scale, track_attendance, binary, likert, categorical, first_workbook_check,
        attendance_patterns, country_summary, profiles, demographics, agenda,
        slots, agenda_coding, learning_content, speaker_composition, objectives,
        decisions, protocol, comparators, traceability, rq1_themes, evidence,
        data_dictionary, questionnaire_2025, questionnaire_2026, questionnaire_common,
        open_comment_inventory, access_process,
        capacity_pressure, scholarship_model, scholarship_rule, speaker_by_track,
        propositions, output_traceability, explanatory_synthesis,
        curricular_connections, corpus_metadata, speaker_detailed,
        networking_design, communication_use, rover_design, expertise_fields, source_schemas, transition_nodes, transition_edges,
        paper_objectives, paper_corpus, paper_participation, rq_document,
    )

    (TABLES / "respondent-indicators.tex").write_text(generate_respondent_table(binary), encoding="utf-8")
    (TABLES / "evidence-by-rq.tex").write_text(generate_evidence_table(evidence), encoding="utf-8")
    (TABLES / "objective-evolution.tex").write_text(generate_traceability_table(traceability), encoding="utf-8")
    (TABLES / "longitudinal-comparators.tex").write_text(generate_comparator_table(comparators), encoding="utf-8")
    (TABLES / "administrative-layers-audit.tex").write_text(generate_administrative_table(admin), encoding="utf-8")
    (TABLES / "track-attendance.tex").write_text(generate_track_attendance_table(track_attendance), encoding="utf-8")
    (TABLES / "educational-objectives.tex").write_text(generate_objectives_table(objectives), encoding="utf-8")
    (TABLES / "design-decisions.tex").write_text(generate_decisions_table(decisions), encoding="utf-8")
    (TABLES / "future-evaluation-protocol.tex").write_text(generate_protocol_table(protocol), encoding="utf-8")
    (FIGURES / "learning-ecosystem.tex").write_text(generate_learning_tikz(transition_nodes), encoding="utf-8")
    (FIGURES / "questionnaire-summary.tex").write_text(
        generate_questionnaire_tikz(questionnaire_2025, questionnaire_2026),
        encoding="utf-8",
    )
    (FIGURES / "scale-trajectory.tex").write_text(generate_scale_tikz(scale), encoding="utf-8")
    (TABLES / "access-process.tex").write_text(generate_access_process_table(access_process), encoding="utf-8")
    (TABLES / "capacity-pressure.tex").write_text(generate_capacity_pressure_table(capacity_pressure), encoding="utf-8")
    (TABLES / "scholarship-selection-model.tex").write_text(generate_scholarship_model_table(scholarship_model), encoding="utf-8")
    (TABLES / "speaker-composition-by-track.tex").write_text(generate_speaker_track_table(speaker_by_track), encoding="utf-8")
    (TABLES / "program-objectives.tex").write_text(generate_program_objectives_paper(paper_objectives), encoding="utf-8")
    (TABLES / "data-corpus.tex").write_text(generate_data_corpus_paper(paper_corpus), encoding="utf-8")
    (TABLES / "space-integration.tex").write_text(generate_curricular_connections_paper(curricular_connections), encoding="utf-8")
    (TABLES / "administrative-layers.tex").write_text(generate_participation_support_paper(paper_participation), encoding="utf-8")
    (TABLES / "rover-systems-lab.tex").write_text(generate_rover_systems_lab_paper(rover_design), encoding="utf-8")
    (TABLES / "explanatory-synthesis.tex").write_text(generate_explanatory_synthesis_paper(explanatory_synthesis), encoding="utf-8")
    (TABLES / "returned-questionnaire-patterns.tex").write_text(generate_questionnaire_summary_paper(questionnaire_common), encoding="utf-8")
    (TABLES / "participant-profiles.tex").write_text(generate_participant_profiles_paper(profiles), encoding="utf-8")
    (TABLES / "track-capacity.tex").write_text(generate_track_capacity_paper(tracks, capacity_pressure), encoding="utf-8")
    (TABLES / "track-sequence-demand.tex").write_text(generate_track_sequence_demand_paper(track_combinations), encoding="utf-8")
    (TABLES / "proposition-evidence-map.tex").write_text(generate_proposition_map_table(propositions), encoding="utf-8")
    (FIGURES / "learning-ecosystem.svg").write_text(svg_learning_ecosystem(learning_content), encoding="utf-8")
    (FIGURES / "program-transition.svg").write_text(svg_program_transition(transition_nodes, transition_edges), encoding="utf-8")
    (FIGURES / "theory-practice-distributions.svg").write_text(svg_theory_practice(questionnaire_2025, questionnaire_2026), encoding="utf-8")
    for stale_output in (
        TABLES / "learning-ecosystem.tex",
        TABLES / "redesign-traceability.tex",
        TABLES / "curricular-connections.tex",
        TABLES / "participation-support.tex",
        TABLES / "track-demand.tex",
        TABLES / "speaker-composition.tex",
        TABLES / "refined-lessons.tex",
        TABLES / "longitudinal-questionnaire.tex",
        FIGURES / "track-selection.svg",
        FIGURES / "speaker-composition.svg",
        FIGURES / "evidence-coverage.svg",
    ):
        if stale_output.exists():
            stale_output.unlink()
    audit_only_dir = RESULTS / "audit-only"
    if audit_only_dir.exists():
        for stale_file in audit_only_dir.rglob("*"):
            if stale_file.is_file():
                stale_file.unlink()
        for stale_dir in sorted((path for path in audit_only_dir.rglob("*") if path.is_dir()), reverse=True):
            stale_dir.rmdir()
        audit_only_dir.rmdir()
    forbidden_questionnaire_paths = [DATA / "single-track-practice-distributions.csv"]
    audit.check(
        "unsupported track questionnaire artifacts absent",
        not audit_only_dir.exists() and all(not path.exists() for path in forbidden_questionnaire_paths),
        "ratings are not attributed to tracks selected in a multi-select background item",
    )
    (RESULTS / "results-summary.json").write_text(
        json.dumps(results_summary_payload(admin, tracks, slots, questionnaire_common), indent=2) + "\n",
        encoding="utf-8",
    )
    (RESULTS / "validation-report.json").write_text(
        json.dumps(validation_payload(audit), indent=2) + "\n",
        encoding="utf-8",
    )
    for stale_markdown in (RESULTS / "results-summary.md", RESULTS / "validation-report.md"):
        if stale_markdown.exists():
            stale_markdown.unlink()

    print(f"Analysis and audit build: {'PASS' if audit.passed else 'FAIL'} ({sum(ok for _, ok, _ in audit.items)}/{len(audit.items)} checks)")
    print(f"Results: {RESULTS}")
    return 0 if audit.passed else 1


if __name__ == "__main__":
    sys.exit(main())
