{
  "updated": "2026-09-09",
  "tasks": [
    {
      "id": "T01",
      "title": "Recover a reproducible set of assets",
      "priority": "P0",
      "lane": "ready",
      "deps": [],
      "effort": "CPU audit + short checkpoint smoke",
      "question": "Can another run recover the exact representations and outputs used by the first comparison?",
      "why": "The archive contains useful trained models, but a file inventory alone does not establish loadability. Whole-z and c_pool dictionaries answer different questions, and some smaller whole-z checkpoint paths are missing.",
      "steps": [
        "Select one TRACE organism, one c_pool dictionary, one whole-z dictionary and the retained SONAR encoder/decoder. Do not attempt to validate every archived checkpoint at once.",
        "Record model names, checkpoint paths and hashes, dimensions, normalization, tokenizer, package versions, seeds, dataset provenance and split identifiers. Resolve paths inside the chosen environment rather than copying historical absolute paths.",
        "Load each selected artifact and reproduce a saved reference prediction or reconstruction. Check tensor shapes, finite activations and the expected representation type.",
        "Write a portable manifest with an explicit unavailable/excluded entry for anything that cannot be recovered. Link each reproduced output to its source result."
      ],
      "controls": "A successful deserialization is insufficient. Compare a known reference output and a deliberately mismatched configuration; the latter must fail clearly.",
      "measures": "Reference-output agreement, numerical tolerance, checkpoint coverage, missing assets and peak memory for the smoke run.",
      "deliverable": "asset_manifest.json, pinned environment notes, and smoke_results.json with a runnable command.",
      "gate": "Start the benchmark comparison only with artifacts that reproduce their reference behavior. Missing historical checkpoints are exclusions, not a reason to silently substitute a new model.",
      "first": "Read 4l-inventory.json and choose the four representative assets. Estimate the cost from their actual file sizes before loading."
    },
    {
      "id": "T02",
      "title": "Reconcile the scientific claims",
      "priority": "P0",
      "lane": "ready",
      "deps": [],
      "effort": "Reading + cached result analysis",
      "question": "Which claims still hold once the earlier FABLE positives, later H1–H3 checks and corrected benchmark baselines are considered together?",
      "why": "The four source pages describe different snapshots. Their broadest headlines can disagree even when their underlying task definitions differ.",
      "steps": [
        "Create one row per claim with its exact estimand, substrate, reader family, available query information, vocabulary/construction split, power gate and independent supporting result.",
        "Map the existing query-conditioned reader and decoder-parser positives against H1. Determine whether each difference is a changed task, changed access, failed instrument or a genuine contradiction.",
        "Reconcile W40 matched-topic results using the same valid subset and denominator. Mark undertrained reconstruction results and keep c_pool separate from whole-z.",
        "Update the proposed abstract and board wording: separate gross from subtle fabrication, post-hoc capacity estimates from frozen tests, and observational geometry from causal localization.",
        "Only then rerun the historical evidence ledger with the corrected support/refutation inputs. Preserve the old version and label the aggregation heuristic."
      ],
      "controls": "Inspect primary result JSONs and code for each disputed number. Do not resolve conflicts by selecting the newest report or the highest confidence label.",
      "measures": "Claims with traceable evidence, unresolved contradictions, instrument failures, corrections and remaining replication requirements.",
      "deliverable": "claims_reconciled.csv plus a concise change log and proposed paper/board wording.",
      "gate": "No claim is formally promoted simply because bookkeeping is complete. Every promoted claim needs its stated evidence and power requirements.",
      "first": "Start with binding, the 460-bit estimate, the 16% atom-energy claim and the corrected TRACE premium."
    },
    {
      "id": "T03",
      "title": "Freeze and validate the benchmark",
      "priority": "P1",
      "lane": "dependent",
      "deps": [
        "T01"
      ],
      "effort": "CPU controls + small inference pilot",
      "question": "Can the evaluation distinguish a useful interpretation from a plausible story or a surface shortcut?",
      "why": "TRACE previously rewarded unequal input access. A benchmark where the best permitted method must tie a strong baseline cannot measure the progress we want.",
      "steps": [
        "Specify three tracks separately: unseen activation prediction, unseen intervention-outcome prediction, and practical uplift from latent access under a matched budget.",
        "Choose a small TRACE organism set. Use GAUGE to validate the measurement instrument and SIEVE/TAE-Bench components for stimulus generation and encoder capability checks.",
        "Freeze information access, training/validation/test boundaries, the unit of independence, the primary score, a minimum useful effect, and the stopping rule before running candidate methods.",
        "Build privileged positive controls and strong text-only or surface-mining baselines. Include corrupted explanations, shuffled labels, random directions and no-op interventions.",
        "Run a small control pilot to estimate variance; choose sample size using the desired effect and independent clusters. Keep a fresh private variant for final evaluation."
      ],
      "controls": "Match text access, number of decoder calls and explanation/query budgets. Surface baselines should include subword cues. A corrupted description should score worse on the concept it was supposed to explain.",
      "measures": "Primary prediction loss or calibrated accuracy, premium over the strongest baseline, family/model-cluster intervals, control separation and total inference cost.",
      "deliverable": "benchmark_contract.md, versioned splits, positive/null control outputs and one comparison-ready evaluator.",
      "gate": "If the privileged control cannot beat the baseline at the declared effect, or a corrupted method passes, repair the task/metric before comparing interpretation methods.",
      "first": "Choose one retained organism where an internally informed reader has an explicitly testable advantage under the proposed access rules."
    },
    {
      "id": "T04",
      "title": "Compare automated explanation methods",
      "priority": "P1",
      "lane": "dependent",
      "deps": [
        "T03",
        "T06"
      ],
      "effort": "Bounded explanation and decoder inference",
      "question": "Does an explanation predict new activations and intervention effects better than matched baselines?",
      "why": "Feature names and detection scores can look good even when the feature produces repetitive or semantically destructive outputs.",
      "steps": [
        "Sample a stratified pilot of frequent/rare, lexical/paraphrase-stable and steerable/non-steerable features, plus known negatives. Treat 100–200 features as a pilot suggestion until power is measured.",
        "Implement four arms: activation-example descriptions, intervention/output descriptions, a combined method, and an adaptive hypothesis-testing agent. Reuse existing flipbook machinery.",
        "Give the adaptive arm the same total query/cost budget as a nonadaptive search arm. Record all examples shown to explainers and keep evaluation examples unseen.",
        "Require a structured prediction, applicable domain, uncertainty and anticipated collateral effects from every explanation.",
        "Evaluate once on the frozen split, inspect failures with blinded annotation, then confirm any useful advantage on another trained substrate."
      ],
      "controls": "Text-only predictors, random/PCA directions, corrupted descriptions, shuffled feature-description assignments and repeated seeds. Select features without inspecting their final test score.",
      "measures": "Prediction loss, confident false claims, coverage, intervention success, collateral changes, fluency and cost. Report these separately rather than hiding tradeoffs in one score.",
      "deliverable": "method_comparison.csv, all explanation/prediction records, uncertainty estimates and a short failure atlas.",
      "gate": "A gain must exceed the minimum useful effect and survive independent replication. A fluent description or a point estimate without adequate power is exploratory.",
      "first": "Implement the activation-only and output-only baselines on a small frozen feature list before adding an adaptive agent."
    },
    {
      "id": "T05",
      "title": "Extend the multilingual pivot experiment",
      "priority": "P1",
      "lane": "ready",
      "deps": [],
      "effort": "Translation curation + short decoding; CPU maps later",
      "question": "What information changes when a French sentence vector is decoded directly into Chinese or passed through English text or an English-reference vector map?",
      "why": "The completed 24-item pilot demonstrates direct translation, but route agreement and cosine do not identify the internal language of a vector. A greedy omission disappeared with beam search.",
      "steps": [
        "Review the saved greedy and beam-5 generations. Annotate paraphrase differences separately from unsupported details, omissions, role changes and corrupted tokens.",
        "Construct an independently checked parallel dataset with contextualized gender, formal address, number, aspect, negation and role contrasts. Resolve singular/plural vous explicitly and allow valid target-language alternatives.",
        "Keep direct decoding and French/English/Chinese text pivots. Add another non-English pivot before treating English as special. Freeze decoding settings and report sensitivity separately.",
        "On a sufficiently large disjoint training set, fit identity, mean-offset, orthogonal Procrustes and regularized linear French-to-English-reference vector maps. Compare a French-to-Chinese-reference map too.",
        "Tune only on validation; hold out meanings and templates. Decode mapped vectors into Chinese and score semantics independently of SONAR, alongside geometric alignment.",
        "If pursuing an internal-computation claim, add layerwise language/meaning readouts and controlled interventions. This is a separate follow-up to the behavioral translation test."
      ],
      "controls": "The identity map is mandatory. An extra same-language round trip controls for repeated compression. A shuffled-pair map and norm-matched perturbation help separate learned alignment from arbitrary vector damage.",
      "measures": "Meaning preservation by phenomenon, unsupported specificity, omissions, corruption, route consistency, paired-versus-contrast geometry and uncertainty clustered by contrast family.",
      "deliverable": "validated_parallel_items.jsonl, annotation rubric, route_comparison.csv, map fit/validation records and held-out generations.",
      "gate": "Do not fit a 1024-D transform on 24 examples and call it generalization. Do not identify English as an internal pivot from cosine or language identification alone.",
      "first": "Inspect the 24-item pilot, especially tu/vous, sister-age specification, bicycle specificity and the Chinese apple-token corruption."
    },
    {
      "id": "T06",
      "title": "Calibrate SAE inference and compare recipes",
      "priority": "P1",
      "lane": "dependent",
      "deps": [
        "T01"
      ],
      "effort": "CPU calibration + a bounded GPU training comparison",
      "question": "Does a faithful Matryoshka recipe yield more useful features than plain BatchTopK at comparable cost and sparsity?",
      "why": "The local BatchTopK evaluation uses batch selection, so an example can change activations when its batch companions change. The old Matryoshka trial also differs from the reference training recipe.",
      "steps": [
        "Preserve the existing code/results and add a calibrated fixed inference threshold using a separate calibration split.",
        "Check the same examples alone and with different batch companions. Report held-out realized L0, its distribution, FVU, feature frequencies and dead features.",
        "Audit the Matryoshka auxiliary loss and relative weighting of prefix losses. Document every intentional deviation from the recipe.",
        "At one retained width and sparsity, compare plain BatchTopK and Matryoshka+BatchTopK using the same data, normalization and at least two seeds. Record training and inference cost.",
        "Run the T04 evaluation plus paraphrase consistency, hierarchy/absorption controls and semantic intervention checks. Use retained seed pairs for an ensemble pilot only after this baseline is trustworthy."
      ],
      "controls": "Batch-companion invariance, held-out calibration, matched realized sparsity and reconstruction, seed variation and random/PCA directions. Match total width and active-feature budget for ensembles.",
      "measures": "FVU and L0 distributions as diagnostics; feature-prediction and causal semantic quality as the deciding outcomes, with cost and uncertainty.",
      "deliverable": "versioned inference calibration, recipe audit, two-seed comparator artifacts and a feature-quality table.",
      "gate": "Do not expand width because an automated naming score improves. Escalate only after a held-out feature-quality gain or an informative diagnosis of the current failure.",
      "first": "Run the retained SAE on identical examples with different batch companions to record the current failure before changing inference."
    },
    {
      "id": "T07",
      "title": "Repair binding measurement before training",
      "priority": "P2",
      "lane": "dependent",
      "deps": [
        "T02"
      ],
      "effort": "Cached embeddings + calibrated probe fits",
      "question": "Which reader can recover which role information under real vocabulary and construction transfer?",
      "why": "An unconditional global direction, an entity-conditioned reader and a decoder parser have different information and computational access. Their results must not be collapsed into a single binding verdict.",
      "steps": [
        "Recover the earlier conditional-reader and decoder-parser code and evaluate both on identical items and splits alongside H1 readers.",
        "Create additive and structured role/filler rotation or tensor-product positive controls under matched distractors and normalization.",
        "Calibrate power separately for each reader and each encoder. Rotate focal identities and balance labels without alphabetical shortcuts.",
        "Compare unconditioned, query-conditioned, nonlinear and decode-then-parse readers with equivalent target definitions.",
        "Only when the assay passes, compare reconstruction and relational objectives at matched data/capacity. Verify a candidate teacher before retrying distillation."
      ],
      "controls": "Lexical and construction holdout, label permutation, simple surface readers, multiple planted amplitudes, and positive controls suited to the hypothesized code.",
      "measures": "Transfer accuracy/AUC, calibration, minimum detectable effect, sample size and confidence intervals per reader/encoder; reconstruction cost of any objective intervention.",
      "deliverable": "reader_task_matrix.csv, structured-control generator, power curves and a narrowly scoped binding conclusion.",
      "gate": "If a positive-control gate fails, report instrument failure. It cannot support an absence claim. A positive decoder result is not evidence of a universal linear role axis.",
      "first": "Make the exact task-definition matrix before fitting any new probe."
    },
    {
      "id": "T08",
      "title": "Separate decoder failures from monitor failures",
      "priority": "P2",
      "lane": "dependent",
      "deps": [
        "T01",
        "T02"
      ],
      "effort": "Small decoder sweeps + held-out calibration",
      "question": "Can a monitor abstain on semantic errors at useful coverage, and how much does decoder training change the answer?",
      "why": "Round-trip cosine detected some gross perturbations, but a threshold fitted on the same sweep has not established transfer to subtle, fluent meaning changes.",
      "steps": [
        "Recover H2 operators and judged examples. Verify that a candidate noise-robust decoder checkpoint is actually accessible and compatible before scheduling its comparison.",
        "Prepare clean, grossly perturbed and near-manifold meaning-change sets; include natural text and role/negation edits with full proposition annotations.",
        "Compare base and robust decoders at matched decoding settings. Include no-op, random norm-matched and mean/shuffled-latent controls to reveal decoder-prior effects.",
        "Fit abstention rules on calibration data only. Evaluate risk versus coverage on held-out examples and shifted domains.",
        "Audit unsupported details and off-target propositions, rather than checking only the attribute deliberately edited."
      ],
      "controls": "Independent semantic labels, separate calibration/test data, latent-ablated decoder priors and same-dose random directions. Any guarantee requires its actual statistical assumptions.",
      "measures": "Semantic error among accepted outputs, coverage, false rejection of valid paraphrases, intervention success, collateral changes and confidence intervals.",
      "deliverable": "decoder-by-perturbation-by-monitor table, risk/coverage curves and inspectable accepted-error examples.",
      "gate": "A cosine threshold is a diagnostic, not a universal safety certificate. If the robust decoder is unavailable, retain that comparison as an explicit missing arm.",
      "first": "Split the saved H2 examples by independent source before fitting a new threshold."
    },
    {
      "id": "T09",
      "title": "Evaluate an accessible second autoencoder",
      "priority": "P2",
      "lane": "ready",
      "deps": [],
      "effort": "Model download + reference checks + small GPU pilot",
      "question": "Do the findings and interpretation methods survive on an independently built, usable sentence autoencoder?",
      "why": "A second substrate can challenge SONAR-specific findings. Public weights alone do not establish correct installation, good reconstruction or multilingual support.",
      "steps": [
        "Start with QwenAR for reconstruction; evaluate LatentSeal next for its smaller bottleneck and robustness objective. Keep environments separate from the retained SONAR stack.",
        "Pin code and checkpoints and reproduce each release’s own reference checks. For QwenAR, honor its Transformers version requirement and diagnose the checkpoint before fresh sentences.",
        "Run the same unseen English length/domain, role, negation, noise and editing examples through encode-to-one-vector-to-decode. Check that original text never bypasses the vector.",
        "Measure latency, peak memory, reconstruction and semantic retention. Evaluate each model at its own supported dimensions; do not reuse SONAR SAE coordinates.",
        "Choose one competent model for T04 replication. Treat multilingual decoding as a separately gated capability rather than assuming it from a multilingual base model."
      ],
      "controls": "Release-reference reproduction, sentence-alone versus batch checks, fresh text, shuffled/no-op latents and a shared semantic evaluation set.",
      "measures": "Exact reconstruction and semantic fidelity by length/domain, corruption, intervention collateral effects, inference cost and memory.",
      "deliverable": "candidate_capabilities.csv, reference check logs, generated examples and a justified substrate choice.",
      "gate": "Exclude a broken installation from scientific comparisons. If reconstruction is inadequate, it can be a failure case but not a fair positive-quality challenger.",
      "first": "Inspect QwenAR’s released evaluation examples and environment requirements; reproduce those before benchmarking new text."
    },
    {
      "id": "T10",
      "title": "Run only claim-changing follow-up controls",
      "priority": "P2",
      "lane": "dependent",
      "deps": [
        "T02"
      ],
      "effort": "Mostly cached analysis; targeted decoding when needed",
      "question": "Which remaining uncertainty could change a central paper claim with the least new work?",
      "why": "The campaign flagged many possible follow-ups. Their existence does not justify another broad campaign or a list of disconnected measurements.",
      "steps": [
        "Rank the remaining claims by how much a plausible counterresult would change the paper and by the smallest experiment that resolves it.",
        "For operator algebra, test held-out constructions, composition order, inverses and collateral changes at the same dose against random directions.",
        "For capacity, freeze the distortion definition and decoder/inverter; use new length curves and handle censored knees explicitly. Keep the historical estimate post-hoc.",
        "For geometry, measure anisotropy and its expected similarity floor directly. Avoid treating an imported theoretical floor as an empirical mechanism.",
        "For dictionaries, reuse matched-frequency/topic subsets and retained seed pairs. Compare stable subspaces as well as one-to-one feature matches."
      ],
      "controls": "Matched subsets and denominators, fixed decoders, held-out data, null directions and uncertainty over the actual independent units.",
      "measures": "A per-claim primary outcome and decision rule; report negative controls and effects on the original headline.",
      "deliverable": "ranked_followups.md and one compact result per selected crux; explicitly defer the rest.",
      "gate": "Do not run a measurement unless its plausible outcomes would alter a claim, method choice or deployment decision.",
      "first": "Select one operator, capacity, geometry or dictionary crux after T02; write its decision rule before executing."
    },
    {
      "id": "T11",
      "title": "Finish a paper with defensible scope",
      "priority": "P2",
      "lane": "dependent",
      "deps": [
        "T02"
      ],
      "effort": "Writing + reproducible plotting",
      "question": "What is the strongest coherent contribution supported by the reconciled evidence?",
      "why": "The draft has substantial body text but missing publication figures and related work. Its finished-section labels should not conceal unresolved scientific scope.",
      "steps": [
        "Rewrite the abstract around the final scoped contributions and their limitations. Keep SONAR observations separate from ladder-only results.",
        "Add primary related work on sentence-role probing, ROLE, conditional probing, SONAR/LCM and automated interpretation before asserting novelty.",
        "Generate figures from provenance-linked data: binding plus power, capacity with scope/censoring, editing doses, auditing risk/coverage, dictionary stability and benchmark method comparison as results become available.",
        "Include instrument failures, retractions and negative results in the evidence table. Label human-proxy judgments as proxies.",
        "Perform a cross-document pass so paper, plan, board and release describe the same statuses."
      ],
      "controls": "Trace every plotted number to a file and computation; check that figure captions state the relevant split, reader, substrate and uncertainty.",
      "measures": "Citation coverage, reproducible figures, resolved contradictions and explicit missing experiments; not the heuristic ledger posterior.",
      "deliverable": "updated paper source, figure scripts/artifacts and a claim-to-evidence appendix.",
      "gate": "Writing can proceed after claim reconciliation. New experimental sections remain provisional until their own task gates pass.",
      "first": "Draft the related-work outline and a six-figure evidence map using only results that survive T02."
    },
    {
      "id": "T12",
      "title": "Package a reproducible release candidate",
      "priority": "P2",
      "lane": "dependent",
      "deps": [
        "T03",
        "T11"
      ],
      "effort": "Packaging + fresh-checkout smoke",
      "question": "Can someone reproduce the benchmark contract and its controls without the original workspace?",
      "why": "The archive is rich but has hard-coded paths, mixed historical snapshots and packages that have not been validated as a public release.",
      "steps": [
        "Create a minimal release layout with data provenance, model requirements, licenses, pinned environments and accessible small examples.",
        "Replace workspace-specific paths with documented configuration. Package control data and score definitions alongside the implementation.",
        "Provide a CPU instrument smoke and a small GPU model smoke, including one known-valid result and one intentional instrument failure.",
        "Validate the candidate in an isolated fresh checkout and record time, dependencies and downloaded asset sizes.",
        "Prepare release notes stating supported substrates, known limitations and reproduction commands. Keep public publishing as a separate explicitly executed action."
      ],
      "controls": "Fresh checkout, no implicit cached files, reproducible expected outputs and an instrument-failure example that fails clearly.",
      "measures": "Smoke reproducibility, missing undeclared dependencies, portability and agreement with the documented benchmark contract.",
      "deliverable": "release candidate directory, licenses/provenance, reproducibility log and release notes.",
      "gate": "Packaging success does not promote the underlying scientific claims. Do not label the artifact publicly released until it actually is.",
      "first": "List the smallest files needed to run T03’s positive and null controls without the rest of the campaign archive."
    },
    {
      "id": "D01",
      "title": "Consolidate the four project reports",
      "priority": "Done",
      "lane": "done",
      "deps": [],
      "summary": "Combined the roundup, paper, literature review and campaign board with the status audit, SAE review and 4l inventory. Scientific claims still require T02 reconciliation.",
      "href": "#background"
    },
    {
      "id": "D02",
      "title": "Run the multilingual pilot",
      "priority": "Done",
      "lane": "done",
      "deps": [],
      "summary": "24 French/English/Chinese triples; nine direct routes and three French pivot routes, with greedy decoding and beam size 5. No fitted vector map or independent bilingual scoring yet.",
      "href": "multilingual.html"
    }
  ]
}
