{
  "schema_version": "1.0",
  "use_cases": [
    {
      "id": "use-case-brca1-brca2-germline-interpretation",
      "slug": "brca1-brca2-germline-interpretation",
      "title": "Interpret BRCA1/BRCA2 germline variants",
      "question": "Which evidence-support methods improve BRCA1/BRCA2 variant review without increasing serious classification errors?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "hereditary cancer",
        "germline",
        "BRCA1",
        "BRCA2",
        "ENIGMA",
        "variant classification",
        "VUS",
        "ACMG AMP"
      ],
      "intended_users": [
        "Germline variant scientists",
        "Hereditary-cancer genetics reviewers"
      ],
      "decision": "Choose methods that assemble and assess variant evidence under gene-specific criteria for a professional BRCA1/BRCA2 classification review.",
      "inputs": [
        "Confirmed germline variant, reference transcript and assay limitations",
        "Population frequency, segregation, phenotype and functional evidence with dates",
        "A relevant, versioned ClinGen ENIGMA specification and evidence cutoff"
      ],
      "output": "A classification dossier with criterion-level evidence, conflicts, uncertainty and further information needed for review.",
      "setting": "Germline BRCA1/BRCA2 interpretation in hereditary-cancer testing. Additional genes require their own specifications and comparisons.",
      "exclusions": [
        "Germline classification does not estimate an individual's absolute cancer risk or select treatment.",
        "Tumour-only sequencing does not confirm germline status.",
        "Functional scores and computational predictions are evidence inputs, not complete classifications."
      ],
      "clinical_scope": "Clinical research on hereditary-cancer evidence review. A VUS does not justify management changes; a negative panel does not remove risk associated with family history.",
      "evidence_gaps": [
        "Independent complete clinical classifications, serious-error adjudication and equal-evidence reviewer-time comparison were not found in selected evidence.",
        "BayesDel labels derive from functional assays; threshold calibration is not held-out pathogenicity validation.",
        "Study v1.0 manual pilot revises specifications on the same variants; current specification version must be pinned for any new run.",
        "Generic BayesDel thresholds from Pejaver are a relevant alternative described in Table S4; supplement download returned challenge HTML, so the four primary-body percentages/bounds are entered while unverified Table S4 cells remain missing.",
        "Qualified human scientific review remains outstanding; automated source transcription does not establish experimental replication."
      ],
      "citations": [
        {
          "source_id": "use-case-source-clinical-priorities-2026-09-28",
          "locator": "C3 — BRCA1/BRCA2 germline interpretation"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Can an evidence-support method reduce review effort without increasing serious classification or criteria errors relative to standard gene-specific review?",
        "baselines": [
          "Manual criteria-based review using the same evidence cutoff and ClinGen ENIGMA specification"
        ],
        "outcomes": [
          "Serious classification disagreements and criterion-assignment errors",
          "Coverage of unresolved cases and preservation of conflicts",
          "Review time against independent expert adjudication"
        ],
        "validation_requirements": [
          "Record the specification version, classification date and germline confirmation.",
          "Audit predictor circularity and overlap with functional training data.",
          "Use temporal and gene/domain separation appropriate to the intended claim.",
          "Retain uncertain and conflicting cases and predefine error severity.",
          "Obtain blinded independent adjudication by qualified hereditary-cancer reviewers."
        ],
        "next_step": "Close the remaining endpoint, independence and comparator gaps documented in the 30 September 2026 audit and linked article issue; obtain qualified scientific review."
      },
      "planned_work": []
    },
    {
      "id": "use-case-cell-type-annotation-transfer",
      "slug": "cell-type-annotation-transfer",
      "title": "Transfer cell-type annotations to a new dataset",
      "question": "Which annotation workflow can label my new dataset reliably and recognise unsupported cell populations?",
      "area": "cells-tissues",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "cell-type annotation",
        "reference transfer",
        "label transfer",
        "single-cell",
        "single-nucleus",
        "cell ontology",
        "rare cells",
        "unknown cell types",
        "Azimuth",
        "CellTypist",
        "scANVI",
        "scTab"
      ],
      "intended_users": [
        "Single-cell analysts",
        "Cell atlas researchers"
      ],
      "decision": "Choose an annotation workflow and reference, accept supported labels, and identify cells that require expert review or additional measurements.",
      "inputs": [
        "Query count data with tissue, disease, donor and assay metadata",
        "A versioned reference and a defined cell-label hierarchy",
        "Independent marker, protein or expert evidence where available",
        "Requirements for coverage, uncertainty and resource use"
      ],
      "output": "Cell-type labels at the requested ontology level, confidence estimates and an explicit unassigned group.",
      "setting": "Research annotation of a new single-cell or single-nucleus cohort. Transfer across donors, studies, disease contexts and technologies is assessed separately.",
      "exclusions": [
        "Atlas labels are not infallible ground truth.",
        "Embedding separation and batch mixing do not establish correct cell identity.",
        "Annotation, batch integration and discovery of new disease states require separate evaluations."
      ],
      "clinical_scope": "Research only. Annotation accuracy does not establish the validity of a clinical diagnostic classifier.",
      "evidence_gaps": [
        "Automated source review only; independent human scientific review remains outstanding.",
        "Checkpoint and split hashes unextracted; no deployment calibration claim.",
        "Reference labels are author annotations, not independent ground truth.",
        "Unknown-type rejection ROC appears in Supplementary Figure 4 but numerical curve labels are not extracted here."
      ],
      "citations": [
        {
          "source_id": "use-case-source-research-priorities-2026-09-28",
          "locator": "R4 — Cell-type annotation transfer"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Does a method improve known-type annotation and unknown-type recognition at matched coverage over conventional methods across independent studies?",
        "baselines": [
          "Marker-based and nearest-reference annotation",
          "Appropriate pinned Azimuth, CellTypist or scANVI workflows",
          "Embeddings with a simple classifier where applicable"
        ],
        "outcomes": [
          "Class-specific and hierarchy-aware errors",
          "Rare-type and absent-reference performance",
          "Calibration and accuracy as uncertain cells are left unassigned",
          "Resource use and transfer across studies and platforms"
        ],
        "validation_requirements": [
          "Hold out donors, whole studies and platforms as required by the intended query.",
          "Audit duplicated cells, donors and labels across references and pretraining data.",
          "Use independent expert or orthogonal label checks and preserve uncertainty in reference labels.",
          "Match ontology level and coverage; keep integration and disease-state discovery separate."
        ],
        "next_step": "Close the remaining endpoint, independence and comparator gaps documented in the 30 September 2026 audit and linked article issue; obtain qualified scientific review."
      },
      "planned_work": []
    },
    {
      "id": "use-case-cnv-detection-characterisation",
      "slug": "cnv-detection-characterisation",
      "title": "Select a copy-number variant detection and characterisation workflow",
      "question": "Which workflow detects and characterises copy-number changes accurately and helps an analyst assess the evidence?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "cnv detection characterisation"
      ],
      "intended_users": [
        "Clinical researchers scoping the available diagnostic-genomics evidence",
        "Computational researchers comparing exact evaluated configurations"
      ],
      "decision": "Inspect the DRAGEN 4.2 CNV/SV benchmarking F-score evidence for 1-5 kb deletions before selecting a workflow; evidence-visualisation usability is not addressed by this bounded intake.",
      "inputs": [
        "Whole-genome sequencing aligned reads",
        "The size range and variant class (deletion, duplication, etc.) of clinical interest"
      ],
      "output": "A sourced F-score for one size-stratified deletion class against a truth set; no visualisation-usability or clinical-reporting claim.",
      "setting": "DRAGEN 4.2's CNV/SV benchmarking (HG002, GIAB SV truth set) reports an F-score for 1-5 kb deletion detection (selected Table S4 cells, sheet CNVbenchmarking, H6=.926, L6=.391).",
      "exclusions": [
        "Duplications, larger/smaller size strata, and tumour CNV (not ingested in this bounded intake)",
        "Evidence-visualisation usability for an analyst (not measured by this F-score endpoint)"
      ],
      "clinical_scope": "Clinical applicability is not established beyond this one size-stratified deletion F-score; a conflicting comparator figure (.391 vs 39.20% in source table/prose) is preserved unresolved and not promoted.",
      "evidence_gaps": [
        "Unreported per-bin counts behind the printed F-score.",
        "No duplication, tumour-CNV or clinical-reporting endpoint is ingested in this bounded intake.",
        "No foundation-model applicability is established for this task."
      ],
      "citations": [
        {
          "source_id": "amp-source-dragen",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        },
        {
          "source_id": "amp-source-dragen-supplementary-tables",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Which workflow detects and characterises copy-number changes accurately and helps an analyst assess the evidence?",
        "baselines": [
          "The conventional/author-introduced workflow measured in the linked primary source(s)."
        ],
        "outcomes": [
          "The declared endpoint in this use case's active mapping(s); see evidence_gaps for what remains open."
        ],
        "validation_requirements": [
          "Independent held-out population matched to the intended clinical setting.",
          "Qualified human scientific review before any clinical-validation claim."
        ],
        "next_step": "A dedicated rewire-benchmarks protocol/run task covering duplications and tumour CNV, plus an analyst-facing visualisation-usability evaluation."
      },
      "planned_work": [
        {
          "title": "Focused benchmark/run task for rewire-benchmark-data issue #17",
          "url": "https://github.com/rewire-bio/rewire-benchmark-data/issues/17",
          "status": "planned",
          "reason": "Reviewed literature evidence is bounded (see evidence_gaps); closing the remaining decision gap needs a dedicated protocol/run task with frozen population and matched controls, tracked in rewire-benchmarks."
        }
      ]
    },
    {
      "id": "use-case-diagnostic-dna-pathogen-identification",
      "slug": "diagnostic-dna-pathogen-identification",
      "title": "Select a DNA pathogen-identification workflow for diagnostic testing",
      "question": "Which DNA sequencing classification workflow detects clinically relevant pathogens and handles contamination or organisms absent from the reference?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "diagnostic dna pathogen identification"
      ],
      "intended_users": [
        "Clinical researchers scoping the available diagnostic-genomics evidence",
        "Computational researchers comparing exact evaluated configurations"
      ],
      "decision": "Inspect the Karius plasma microbial cfDNA positive-percent-agreement evidence against initial blood culture before selecting a workflow and reference standard for the intended specimen population.",
      "inputs": [
        "Plasma microbial cell-free DNA sequencing reads",
        "The reference standard available for agreement/sensitivity comparison (e.g. blood culture, composite adjudication)"
      ],
      "output": "A sourced positive-percent-agreement figure against one reference standard, with the reference standard's own detection limits explicit; no general sensitivity claim against all true infections.",
      "setting": "The Karius 2019 plasma microbial cfDNA sequencing assay achieves 93.7% (59 of 63) positive percent agreement with initial blood culture in a prospective 350-patient sepsis-alert cohort (348 paired results).",
      "exclusions": [
        "Sensitivity against a composite/adjudicated reference standard (a separate, excluded endpoint in the source due to a printed CI conflict)",
        "RNA viral pathogen detection (this is a DNA-only assay)"
      ],
      "clinical_scope": "Clinical applicability is bounded to agreement with initial blood culture, a reference standard that does not identify every true infection; this is not an estimate of sensitivity against all infections in the cohort.",
      "evidence_gaps": [
        "The composite-reference negative-percent-agreement CI is internally conflicting in the source (54.8-70.0 vs 55.2-70.4) and is excluded from this intake.",
        "The exact executable/database version is not pinned in the inspected primary text.",
        "The primary article's public redistribution licence is absent; only a compact factual table excerpt is archived."
      ],
      "citations": [
        {
          "source_id": "amp-20261007-dna-pathogens-source",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Which DNA sequencing classification workflow detects clinically relevant pathogens and handles contamination or organisms absent from the reference?",
        "baselines": [
          "The conventional/author-introduced workflow measured in the linked primary source(s)."
        ],
        "outcomes": [
          "The declared endpoint in this use case's active mapping(s); see evidence_gaps for what remains open."
        ],
        "validation_requirements": [
          "Independent held-out population matched to the intended clinical setting.",
          "Qualified human scientific review before any clinical-validation claim."
        ],
        "next_step": "A dedicated rewire-benchmarks protocol/run task with a prospectively adjudicated composite reference standard, not solely initial blood culture."
      },
      "planned_work": [
        {
          "title": "Focused benchmark/run task for rewire-benchmark-data issue #13",
          "url": "https://github.com/rewire-bio/rewire-benchmark-data/issues/13",
          "status": "planned",
          "reason": "Reviewed literature evidence is bounded (see evidence_gaps); closing the remaining decision gap needs a dedicated protocol/run task with frozen population and matched controls, tracked in rewire-benchmarks."
        }
      ]
    },
    {
      "id": "use-case-diagnostic-genomics-model-execution",
      "slug": "diagnostic-genomics-model-execution",
      "title": "Select an execution workflow for large-scale diagnostic genomics",
      "question": "Which eligible model configuration and execution workflow can meet a diagnostic genomics team's resource, throughput and reproducibility constraints?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "diagnostic genomics model execution"
      ],
      "intended_users": [
        "Clinical researchers scoping the available diagnostic-genomics evidence",
        "Computational researchers comparing exact evaluated configurations"
      ],
      "decision": "Inspect the DRAGEN 4.2 single-sample runtime baseline before scoping a foundation-model execution workflow's resource/throughput budget; this conventional baseline is a proxy, not a model-execution benchmark.",
      "inputs": [
        "The diagnostic genomics team's available hardware configuration",
        "The required per-sample turnaround and batch throughput"
      ],
      "output": "A sourced single-sample total runtime baseline on two hardware configurations; no foundation-model execution benchmark is ingested in this bounded intake.",
      "setting": "DRAGEN 4.2 Phase 4 total single-sample wall-clock runtime for HG002 is 1,838.5 seconds on one hardware configuration and 5,521.21 seconds on an AWS HG002 configuration (f1.4xlarge, Xeon E5-2686v4, 16 threads); these are hardware comparators, not model comparators, and stage times cannot be summed due to concurrency.",
      "exclusions": [
        "Any foundation-model inference runtime or throughput benchmark (none is ingested in this bounded intake)",
        "Peak memory usage (host RAM capacity in the hardware comparator is not measured peak usage)",
        "Clinical-reporting turnaround time"
      ],
      "clinical_scope": "Clinical applicability is not established: this is a conventional variant-calling pipeline's operational runtime baseline, a proxy for the resource/throughput constraints a foundation-model execution workflow would need to meet, not itself a model-execution benchmark.",
      "evidence_gaps": [
        "No peak-memory endpoint is reported.",
        "No runtime repeat/variance (CI) is reported.",
        "A separate ~2-hour joint-aggregation figure over 3,202 genomes at concurrency 200 is a different operation and is not assigned to this single-sample runtime."
      ],
      "citations": [
        {
          "source_id": "amp-source-dragen",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        },
        {
          "source_id": "amp-source-dragen-supplementary-tables",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Which eligible model configuration and execution workflow can meet a diagnostic genomics team's resource, throughput and reproducibility constraints?",
        "baselines": [
          "The conventional/author-introduced workflow measured in the linked primary source(s)."
        ],
        "outcomes": [
          "The declared endpoint in this use case's active mapping(s); see evidence_gaps for what remains open."
        ],
        "validation_requirements": [
          "Independent held-out population matched to the intended clinical setting.",
          "Qualified human scientific review before any clinical-validation claim."
        ],
        "next_step": "A dedicated rewire-benchmarks protocol/run task that actually executes an eligible foundation-model configuration and measures its resource/throughput budget."
      },
      "planned_work": [
        {
          "title": "Focused benchmark/run task for rewire-benchmark-data issue #18",
          "url": "https://github.com/rewire-bio/rewire-benchmark-data/issues/18",
          "status": "planned",
          "reason": "Reviewed literature evidence is bounded (see evidence_gaps); closing the remaining decision gap needs a dedicated protocol/run task with frozen population and matched controls, tracked in rewire-benchmarks."
        }
      ]
    },
    {
      "id": "use-case-diagnostic-rna-pathogen-detection",
      "slug": "diagnostic-rna-pathogen-detection",
      "title": "Select an RNA pathogen-detection workflow for diagnostic testing",
      "question": "Which RNA sequencing workflow detects RNA pathogens reliably in the intended specimen and distinguishes real detections from host/background contamination?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "diagnostic rna pathogen detection"
      ],
      "intended_users": [
        "Clinical researchers scoping the available diagnostic-genomics evidence",
        "Computational researchers comparing exact evaluated configurations"
      ],
      "decision": "Inspect the UCSF respiratory RNA mNGS sensitivity-against-original-testing evidence before selecting a workflow and specimen-prep protocol for the intended respiratory-target population.",
      "inputs": [
        "Respiratory specimen RNA (DNase-treated), reverse-transcribed to cDNA libraries",
        "The original clinical reference assay(s) available for sensitivity comparison"
      ],
      "output": "A sourced sensitivity-against-original-clinical-testing figure for one respiratory mNGS assay; no general RNA-virus-only diagnostic accuracy claim.",
      "setting": "A UCSF respiratory RNA metagenomic next-generation sequencing (mNGS) assay (RNA extraction, DNase treatment, cDNA synthesis; SURPI+ pipeline) achieves 93.6% (103 of 110) sensitivity against original clinical respiratory-virus-panel testing in a residual pre-DTCA mixed respiratory-target cohort (adenovirus transcripts included via transcription).",
      "exclusions": [
        "A composite-PPA figure (98.7%, 110.5/113) that conflicts with the source's own printed numbers and is excluded from this intake",
        "Non-respiratory specimen types"
      ],
      "clinical_scope": "Clinical applicability is bounded to agreement with the original clinical panel on one residual-sample cohort; this is not a mixed DNA/RNA sample-prep comparison and not an RNA-virus-only subgroup score, and is not validated prospectively.",
      "evidence_gaps": [
        "A positive-specimen BAL/swab count disagreement between the source's Results and Methods sections (104+6 vs 103+7) is unresolved and preserved.",
        "No confidence interval is extracted for this sensitivity figure in the inspected text."
      ],
      "citations": [
        {
          "source_id": "amp-20261007-rna-pathogens-source",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Which RNA sequencing workflow detects RNA pathogens reliably in the intended specimen and distinguishes real detections from host/background contamination?",
        "baselines": [
          "The conventional/author-introduced workflow measured in the linked primary source(s)."
        ],
        "outcomes": [
          "The declared endpoint in this use case's active mapping(s); see evidence_gaps for what remains open."
        ],
        "validation_requirements": [
          "Independent held-out population matched to the intended clinical setting.",
          "Qualified human scientific review before any clinical-validation claim."
        ],
        "next_step": "A dedicated rewire-benchmarks protocol/run task with a prospective RNA-pathogen cohort and an adjudicated composite reference standard."
      },
      "planned_work": [
        {
          "title": "Focused benchmark/run task for rewire-benchmark-data issue #14",
          "url": "https://github.com/rewire-bio/rewire-benchmark-data/issues/14",
          "status": "planned",
          "reason": "Reviewed literature evidence is bounded (see evidence_gaps); closing the remaining decision gap needs a dedicated protocol/run task with frozen population and matched controls, tracked in rewire-benchmarks."
        }
      ]
    },
    {
      "id": "use-case-egfr-nsclc-actionability-resistance-evidence",
      "slug": "egfr-nsclc-actionability-resistance-evidence",
      "title": "Review EGFR lung-cancer actionability evidence",
      "question": "Which systems retrieve correctly scoped sensitivity and resistance evidence for advanced EGFR-mutant NSCLC?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "EGFR",
        "NSCLC",
        "non-small-cell lung cancer",
        "molecular tumour board",
        "actionability",
        "resistance",
        "CIViC",
        "evidence retrieval"
      ],
      "intended_users": [
        "Molecular tumour-board scientists",
        "Oncologists and clinical molecular scientists reviewing evidence"
      ],
      "decision": "Choose an evidence-review system that connects tumour findings to relevant sensitivity and resistance studies while exposing contradictions and missing context.",
      "inputs": [
        "Interpreted biomarker, genome build/transcript, histology, stage and co-alterations",
        "Prior therapies, progression and sampling dates, including tissue or plasma source",
        "Jurisdiction, evidence cutoff and a versioned source corpus"
      ],
      "output": "Source-linked evidence assertions with regimen, sensitivity or resistance direction, treatment setting, native evidence tier, contradictions and unresolved information.",
      "setting": "Advanced EGFR-mutant non-small-cell lung cancer, including progression after EGFR-targeted therapy. Treatment history and evidence date are part of every comparison.",
      "exclusions": [
        "Evidence retrieval does not select a treatment or establish individual treatment benefit.",
        "Trial eligibility, response and survival require separate evaluations.",
        "Missing evidence does not establish resistance, and evidence tiers from different systems are not interchangeable."
      ],
      "clinical_scope": "Clinical research on support for qualified molecular-oncology evidence review. Applicability depends on tumour, biomarker, prior therapy, sampling context, jurisdiction and evidence date.",
      "evidence_gaps": [
        "No independently adjudicated advanced EGFR-mutant NSCLC evidence-retrieval benchmark with prior therapy, treatment line, date, jurisdiction and contradictions was established by this bounded search.",
        "CIViC study reports pan-cancer aggregate and predictive direction metrics; no published EGFR-only score was found.",
        "OncoTraj (arXiv:2606.11144) predicts longitudinal resistance and is outside the requested evidence-retrieval endpoint.",
        "Manual versioned lookup plus primary-source review still needs measured equal effort.",
        "Qualified human scientific review remains outstanding; automated source transcription does not establish experimental replication."
      ],
      "citations": [
        {
          "source_id": "use-case-source-clinical-priorities-2026-09-28",
          "locator": "C5 — EGFR-mutant NSCLC actionability and resistance evidence"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Does retrieval-assisted review recover more correctly scoped sensitivity/resistance evidence with fewer unsupported assertions than a versioned lookup/manual-search workflow at equal review effort?",
        "baselines": [
          "Versioned knowledgebase lookup plus conventional manual primary-source review"
        ],
        "outcomes": [
          "Relevant-evidence recall and unsupported-assertion rate",
          "Tumour, treatment-line, direction, date and citation fidelity",
          "Contradiction handling, missing-information handling and review time"
        ],
        "validation_requirements": [
          "Freeze the question, corpus and knowledge cutoff; hold out studies and molecular profiles.",
          "Preserve native evidence tiers and include contradictory or inapplicable evidence.",
          "Audit citation support and retain unknown clinical information.",
          "Use independent qualified molecular-oncology adjudication.",
          "Confirm source-specific benchmarking permissions and separate evidence review from treatment and eligibility outcomes."
        ],
        "next_step": "Close the remaining endpoint, independence and comparator gaps documented in the 30 September 2026 audit and linked article issue; obtain qualified scientific review."
      },
      "planned_work": []
    },
    {
      "id": "use-case-genetic-perturbation-response",
      "slug": "genetic-perturbation-response",
      "title": "Assess models for genetic perturbation experiments",
      "question": "Which prediction methods and controls should I test before using expression predictions to plan genetic perturbation experiments?",
      "area": "cells-tissues",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "genetic perturbation",
        "perturbation response",
        "Perturb-seq",
        "single-cell RNA-seq",
        "scRNA-seq",
        "GEARS",
        "CPA",
        "knowledge graph",
        "Norman2019",
        "no-change control",
        "gene expression"
      ],
      "intended_users": [
        "Researchers designing genetic perturbation screens",
        "Computational biologists evaluating perturbation-response models"
      ],
      "decision": "Choose models and controls for a pilot in the intended cell system before using predictions to select follow-up experiments. The current evidence compares expression prediction; it does not establish experimental hit rates.",
      "inputs": [
        "Single-cell expression measurements from perturbed and unperturbed cells, with perturbation and cell-type labels",
        "A study design with multiple measured perturbations and cells per perturbation; combinations require combination examples during GEARS training"
      ],
      "output": "Predicted post-perturbation expression and changes relative to unperturbed controls, assessed against measured responses.",
      "setting": "Research method assessment anchored to the Norman2019 K562 cell-line comparison in GEARS Supplementary Table 6. Both endpoints below describe the same four evaluated configurations.",
      "exclusions": [
        "Cross-cell-type transfer and bulk-sequencing prediction are outside the cited GEARS usage scope.",
        "Training GEARS only on single-gene perturbations does not support reliable combination prediction.",
        "Expression prediction scores do not establish causal mechanism, fitness effects or successful experimental prioritisation."
      ],
      "clinical_scope": "Research use only. These expression endpoints do not establish diagnostic, treatment-selection or patient-response validity.",
      "evidence_gaps": [
        "The exact Table 6 split manifest, scoring gene subset, scored counts and checkpoint revisions remain unextracted. Do not silently label the MSE endpoint as Top20.",
        "The table does not identify its printed spread as SD, SE or a confidence interval; these scores do not establish a statistically supported winner.",
        "Both endpoints and the repeated No Perturb row are overlapping evidence from one source comparison, not independent replications.",
        "Performance in a new cell system and experimental hit rates need a separate prospective evaluation. Human scientific review remains outstanding.",
        "30 September 2026 audit: A benchmark exists for expression responses, but prospective experiment-selection hit rates and transfer to a new cell system remain uncovered by this mapped comparison.",
        "30 September 2026 audit: Table 6 scoring gene set, checkpoint revisions and exact split manifests remain unextracted; do not infer Top20 MSE or confidence intervals from the printed spreads."
      ],
      "citations": [
        {
          "source_id": "coverage-source-gears-supp",
          "locator": "Supplementary Table 6, printed page 34 / PDF page 35; Supplementary Table 1, PDF page 30; Supplementary Notes 5 and 14 (K562 context)."
        },
        {
          "source_id": "evidence-official-c037e3419c04936262a0",
          "locator": "README.md at f374e43e197b295016d80395d7a54ddb81cc6769, A note on usage (lines 15–19) and Core API Interface (lines 20–54)."
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "planned_work": []
    },
    {
      "id": "use-case-mass-spectrum-molecule-shortlisting",
      "slug": "mass-spectrum-molecule-shortlisting",
      "title": "Shortlist molecular identities from tandem mass spectra",
      "question": "Which retrieval configurations merit validation when molecular formula is unknown and a researcher can examine only a short list of candidate structures?",
      "area": "metabolomics",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "metabolomics",
        "tandem mass spectrometry",
        "MS/MS",
        "molecular identification",
        "candidate retrieval",
        "unknown formula",
        "MassSpecGym",
        "MSAlign",
        "DreaMS",
        "ChemBERTa",
        "MCES",
        "formula split"
      ],
      "intended_users": [
        "Metabolomics researchers shortlisting candidate molecular identities for experimental confirmation",
        "Computational researchers choosing retrieval inputs, candidate pools and validation splits"
      ],
      "decision": "Compare formula-free retrieval configurations within a named split and shortlist size, then validate the intended instrument, chemistry and candidate database before selecting a workflow.",
      "inputs": [
        "MS/MS spectra with metadata sufficient to derive neutral molecular mass",
        "A frozen set of candidate molecular structures; the true identity must be present for the measured retrieval endpoint"
      ],
      "output": "Split-specific Recall@1 and Recall@20 evidence for six exact configurations, plus candidate-pool and domain-shift limits.",
      "setting": "MSAlign paper Table 3, MassSpecGym formula and MCES splits, formula-free inputs. Each source candidate pool contains 256 mass-matched PubChem structures. Formula and MCES splits remain distinct.",
      "exclusions": [
        "De novo molecular generation or cases where the true structure is absent from the candidate set",
        "Formula-conditioned retrieval, including MIST, FLARE and MSAlign+Filter",
        "Pooling split-specific results or transferring scores to different candidate database sizes",
        "Authenticated molecular identification and clinical diagnosis"
      ],
      "clinical_scope": "Research only. Candidate retrieval is not a confirmed molecular identification or evidence of clinical diagnostic performance.",
      "evidence_gaps": [
        "The correct structure must be in a 256-candidate mass-filtered pool. Scores do not transfer automatically to a larger database, a different pool-construction procedure or a missing true structure.",
        "MCES and formula splits impose different distribution shifts. The source authors favour formula splits for their deployment interpretation; that choice does not prove that formula-split performance represents every intended application.",
        "Reported methods are the paper implementations: DeepSets training was extended to 50 epochs. These are not interchangeable with original model-paper scores or current checkpoints.",
        "Exact test denominators, split-manifest hashes, checkpoint hashes and evaluator revisions remain unextracted or unreported in the catalogue. No uncertainty values or new independent reproduction are established.",
        "Candidate recall is not an authenticated metabolite identification, a clinical diagnosis or a calibrated probability. Human domain review remains outstanding.",
        "The four metric and split groups come from one paper and do not constitute independent replications.",
        "30 September 2026 audit: v1 source record labels 2605.19752v1 but uses an unversioned URL that now serves v2; old source bytes/values must remain immutable. v1-specific URL returned HTTP406 during this audit.",
        "30 September 2026 audit: For MCES, the v2 Table3 generic caption says three random splits, but Section5.1 explicitly uses two validation/test-swapped variants; the discrepancy is preserved.",
        "30 September 2026 audit: Benchmark assumes the true structure is present among 256 neutral-mass-matched candidates. Unknown candidates, instrument-specific prospective performance and authenticated identification remain gaps."
      ],
      "citations": [
        {
          "source_id": "evidence-official-760ab2fa8c396aeb796c",
          "locator": "Table 3, MassSpecGym MCES and formula blocks, formula-free columns, R@1 and R@20; Sections 4, 5.1 and 5.2"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "planned_work": []
    },
    {
      "id": "use-case-patient-rna-splicing-validation",
      "slug": "patient-rna-splicing-validation",
      "title": "Select a patient-RNA splicing-variant validation workflow",
      "question": "Which DNA/RNA evidence workflow identifies splice-altering variants for inherited-disorder follow-up when patient RNA and tissue-specific context are available?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "patient rna splicing validation"
      ],
      "intended_users": [
        "Clinical researchers scoping the available diagnostic-genomics evidence",
        "Computational researchers comparing exact evaluated configurations"
      ],
      "decision": "Inspect the FRASER known pathogenic-event recovery curve and the sequence-classification proxy evidence before selecting a workflow and designing patient-RNA follow-up for the intended cohort size.",
      "inputs": [
        "Patient skin-fibroblast (or matched tissue) RNA-seq aligned reads",
        "The number of samples available for a FRASER-style aberrant-splicing cohort"
      ],
      "output": "A sourced recovery curve (fraction of known pathogenic splicing events recovered as cohort size grows) and a separate sequence-classification proxy; no diagnostic-yield claim for unknown cases.",
      "setting": "FRASER, applied to a retrospective Kremer rare-mitochondrial-disorder skin-fibroblast RNA cohort (119 samples/105 individuals, 13 known pathogenic splicing events), recovers a mean 85% (11 of 13) of known events at a reduced 30-sample subsample. A separate, proxy sequence-classification AUC for DNA foundation-model embeddings (Feng et al. 2025, Table 1 Acceptor/Donor) is also linked, scoped as a different endpoint.",
      "exclusions": [
        "Diagnostic yield in patients without a known pathogenic event (this is recovery of already-known events, not discovery)",
        "Full-cohort outlier-gene workload counts (12/7/10 genes/sample), a different population from the 30-sample recovery evaluation and not ingested here",
        "Blinded pathogenicity adjudication or VUS reclassification"
      ],
      "clinical_scope": "Clinical applicability is not established beyond known-event recovery. This does not demonstrate prospective diagnostic yield, blinded adjudication, or general tissue transfer.",
      "evidence_gaps": [
        "No numeric confidence interval is reported for the 85% recovery figure in the inspected main text.",
        "Subsampling repeat count and exact score-aggregation method are unextracted.",
        "The sequence-classification proxy mapping (Feng Table 1) measures AUC on a different population (DNABERT-2/NT benchmark splice-site datasets), not minigene reporter exon inclusion or patient RNA, and must not be pooled with the FRASER recovery figure."
      ],
      "citations": [
        {
          "source_id": "amp-oncology-rna-20261007-source-pmc7822922",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Which DNA/RNA evidence workflow identifies splice-altering variants for inherited-disorder follow-up when patient RNA and tissue-specific context are available?",
        "baselines": [
          "The conventional/author-introduced workflow measured in the linked primary source(s)."
        ],
        "outcomes": [
          "The declared endpoint in this use case's active mapping(s); see evidence_gaps for what remains open."
        ],
        "validation_requirements": [
          "Independent held-out population matched to the intended clinical setting.",
          "Qualified human scientific review before any clinical-validation claim."
        ],
        "next_step": "A dedicated rewire-benchmarks protocol/run task with a prospective, blinded patient-RNA cohort including cases without a pre-known pathogenic event."
      },
      "planned_work": [
        {
          "title": "Focused benchmark/run task for rewire-benchmark-data issue #12",
          "url": "https://github.com/rewire-bio/rewire-benchmark-data/issues/12",
          "status": "planned",
          "reason": "Reviewed literature evidence is bounded (see evidence_gaps); closing the remaining decision gap needs a dedicated protocol/run task with frozen population and matched controls, tracked in rewire-benchmarks."
        }
      ]
    },
    {
      "id": "use-case-phenotype-perturbation-selection",
      "slug": "phenotype-perturbation-selection",
      "title": "Select perturbations for a defined cellular response",
      "question": "Which genetic perturbations should I test to produce a defined cellular response?",
      "area": "cells-tissues",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "perturbation selection",
        "phenotype",
        "genetic screen",
        "CRISPR",
        "Perturb-seq",
        "experimental hit rate",
        "cellular response",
        "unseen perturbation",
        "Virtual Cell Challenge",
        "Systema"
      ],
      "intended_users": [
        "Functional genomics screen designers",
        "Cell biologists planning perturbation experiments"
      ],
      "decision": "Allocate a fixed experimental budget to perturbations most likely to produce the prespecified phenotype.",
      "inputs": [
        "A defined cell system, starting state and target phenotype",
        "Candidate genetic perturbations and a fixed testing budget",
        "Measured perturbations and matched controls with guide, replicate and condition metadata",
        "The intended transfer setting: unseen genes, combinations or a new cellular context"
      ],
      "output": "A ranked experimental shortlist with predicted phenotype effects, coverage and uncertainty about unmeasured conditions.",
      "setting": "Research selection for a defined genetic-perturbation screen. Chemical interventions and combinations need their own protocols. This question concerns phenotype hits rather than expression reconstruction alone.",
      "exclusions": [
        "Transcriptome similarity alone does not establish successful experimental selection.",
        "Generalisation to an unseen gene, combination and cell context must be evaluated separately.",
        "Choosing experiments to maximise information gain is a separate decision."
      ],
      "clinical_scope": "Research only. Predicted cellular responses do not establish patient response or treatment-selection validity.",
      "evidence_gaps": [
        "61 nominations / Results 57 selected targets / README 59 targets / 50 post-QC perturbations need reconciliation; no success fraction calculated.",
        "Automated source review only; independent human scientific review remains outstanding.",
        "Custom ranking AUC is not ROC AUC, prospective hit rate or efficacy.",
        "No matched-budget conventional/random prospective comparison; no antitumor efficacy, rescue or selectivity result in this endpoint.",
        "Original model implementations/checkpoints are not pinned in the intake; no uncertainty printed.",
        "Preprint v1; original and reimplemented ranking scores kept separate. Screen2 ranking population is enriched by original nomination methods."
      ],
      "citations": [
        {
          "source_id": "use-case-source-research-priorities-2026-09-28",
          "locator": "R3 — Phenotype-driven perturbation selection"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Does a model select more perturbations producing a prespecified phenotype than simple controls at the same test budget in the intended held-out setting?",
        "baselines": [
          "Random candidate selection and mean-response/no-change controls",
          "Linear or other simple statistical response models",
          "Nearest measured perturbation"
        ],
        "outcomes": [
          "Experimentally confirmed phenotype hits per test budget",
          "Perturbation-specific expression signal as an intermediate endpoint",
          "Coverage and uncertainty for unseen interventions or contexts"
        ],
        "validation_requirements": [
          "Use the same candidate universe, phenotype definition and budget; specify how tied rankings are sampled.",
          "Separate unseen genes, combinations and cell contexts, holding out biological and experimental units.",
          "Audit guide efficacy, shared controls, batch effects and model training overlap.",
          "Do not choose evaluation genes using held-out responses; keep information-gain design separate."
        ],
        "next_step": "Close the remaining endpoint, independence and comparator gaps documented in the 30 September 2026 audit and linked article issue; obtain qualified scientific review."
      },
      "planned_work": []
    },
    {
      "id": "use-case-plant-promoter-reporters",
      "slug": "plant-promoter-reporters",
      "title": "Compare methods for plant promoter experiments",
      "question": "What evidence supports choosing a sequence model for predicting plant core-promoter strength in the intended reporter assay?",
      "area": "dna-genomes",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "plant promoter",
        "core promoter",
        "promoter strength",
        "synthetic biology",
        "STARR-seq",
        "AgroNT",
        "CNN",
        "maize protoplasts",
        "tobacco leaves",
        "Arabidopsis thaliana",
        "Sorghum bicolor",
        "Zea mays"
      ],
      "intended_users": [
        "Plant synthetic-biology researchers prioritising promoter candidates for reporter experiments",
        "Computational researchers selecting an assay-matched promoter prediction method"
      ],
      "decision": "Choose the matching sequence species and reporter host, then inspect the two evaluated methods and remaining validation needs before prioritising new promoter candidates. Keep each assay and species comparison separate.",
      "inputs": [
        "A 170 bp core-promoter sequence spanning −165 to +5 relative to an annotated transcription start site",
        "The sequence species and intended assay host: maize protoplasts or tobacco leaves"
      ],
      "output": "Assay-specific evidence for AgroNT and the CNN from Jores et al., with exact evaluated configurations and source rows; no general plant-performance ranking.",
      "setting": "The evidence concerns held-out core-promoter sequences from Arabidopsis thaliana, Sorghum bicolor and Zea mays. Their activity was measured by transient STARR-seq reporter assays in maize protoplasts or tobacco leaves, using the original studies’ train/test datasets for the model comparisons.",
      "exclusions": [
        "Endogenous gene expression, stable transformed plants, tissue-specific activity outside the two reporter hosts, plant fitness or crop yield",
        "Terminator strength, promoter–terminator combinations, chromatin accessibility and other AgroNT tasks",
        "Prospective validation of newly designed promoters, or pooling the six assay-by-species conditions into one score"
      ],
      "clinical_scope": "This page concerns plant reporter research. The evidence establishes no clinical application.",
      "evidence_gaps": [
        "All values are author-reported and source checked; no independent experimental reproduction or human scientific review is recorded.",
        "The source tables provide no confidence intervals or run variability. Exact fitted checkpoint hashes, task-specific seeds, scoring denominators and executable split manifests remain unreported or unextracted.",
        "The six comparisons retain their own assay system and sequence species. A maize-protoplast model can be evaluated on sequences from any of the three listed species; the host is not the sequence origin.",
        "Held-out labelled examples do not establish exclusion from AgroNT pretraining or generalisation to an unseen species. Its reference-genome pretraining and supervised assay split answer different leakage questions.",
        "Held-out R² describes assay-strength prediction. It is not top-k candidate hit rate, uncertainty for a new promoter, or proof that a designed promoter will work in a stable plant.",
        "These four task-specific configurations do not establish performance for other AgroNT checkpoints or newer methods. Missing checkpoint and fitting detail limits exact recreation of the reported comparison.",
        "The separately flagged Figure 4 chromatin-accessibility source anomaly is outside this page and supplies no evidence for promoter selection.",
        "30 September 2026 audit: Reporter-host and sequence-species conditions cannot be pooled. Jores pooled-species Fig8 and AgroNT species-specific Fig3e protocols remain separate.",
        "30 September 2026 audit: Prospective experimental promoter evolution is described by Jores; no benchmarked prospective model-to-model design hit-rate or stable-plant endpoint has been extracted.",
        "30 September 2026 audit: Uncertainty, exact test counts and split/checkpoint hashes remain unextracted for the newly added historical comparison."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, lines 2–13, R2 column; pair rows by Species, Model and Type"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21); pretraining data and training (Sec14–15)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "planned_work": []
    },
    {
      "id": "use-case-plasma-ctdna-fragmentomics",
      "slug": "plasma-ctdna-fragmentomics",
      "title": "Select a plasma ctDNA fragmentomics detection workflow",
      "question": "Which fragmentomics workflow detects tumour-derived plasma DNA under realistic low tumour fractions and fixed false-positive constraints?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "plasma ctdna fragmentomics"
      ],
      "intended_users": [
        "Clinical researchers scoping the available diagnostic-genomics evidence",
        "Computational researchers comparing exact evaluated configurations"
      ],
      "decision": "Inspect the DELFI repeated-cross-validation sensitivity-at-fixed-specificity evidence before selecting a workflow and validation design for the intended screening or diagnostic population.",
      "inputs": [
        "Plasma cell-free DNA whole-genome sequencing reads",
        "The target specificity (false-positive) constraint for the intended use"
      ],
      "output": "A sourced sensitivity figure at one reported specificity, from internal cross-validation; no prospective screening-validation claim.",
      "setting": "The DELFI stochastic gradient boosting classifier (GC-corrected total and short fragment coverage across 504 genomic bins, 39 arm Z-scores, mitochondrial representation) achieves 73% (152 of 208) sensitivity at a reported 98% specificity, under repeated 10-fold cross-validation (10 repeats) across 208 cancer patients (seven cancer types) and 215 healthy individuals.",
      "exclusions": [
        "A combined mutation+DELFI configuration (115/126, 91%), a different subset/configuration not ingested here",
        "Tumour-fraction-stratified sensitivity (unreported for this endpoint)"
      ],
      "clinical_scope": "Clinical applicability is not established as prospective screening performance: this is internal repeated cross-validation on one assembled cohort, not a separately held-out validation cohort or a prospective screening trial, and the cohort's clinically identified cancers/healthy comparators differ from an intended screening population.",
      "evidence_gaps": [
        "4 of 215 healthy individuals were misclassified at the source-labelled 98% specificity; the printed specificity is retained rather than recomputed.",
        "cfDNA signal reflects total circulating DNA, not purified ctDNA; tumour-fraction limits are unreported for this endpoint."
      ],
      "citations": [
        {
          "source_id": "amp-20261007-ctdna-fragmentomics-source",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Which fragmentomics workflow detects tumour-derived plasma DNA under realistic low tumour fractions and fixed false-positive constraints?",
        "baselines": [
          "The conventional/author-introduced workflow measured in the linked primary source(s)."
        ],
        "outcomes": [
          "The declared endpoint in this use case's active mapping(s); see evidence_gaps for what remains open."
        ],
        "validation_requirements": [
          "Independent held-out population matched to the intended clinical setting.",
          "Qualified human scientific review before any clinical-validation claim."
        ],
        "next_step": "A dedicated rewire-benchmarks protocol/run task with an independent, prospectively collected held-out validation cohort."
      },
      "planned_work": [
        {
          "title": "Focused benchmark/run task for rewire-benchmark-data issue #15",
          "url": "https://github.com/rewire-bio/rewire-benchmark-data/issues/15",
          "status": "planned",
          "reason": "Reviewed literature evidence is bounded (see evidence_gaps); closing the remaining decision gap needs a dedicated protocol/run task with frozen population and matched controls, tracked in rewire-benchmarks."
        }
      ]
    },
    {
      "id": "use-case-plasma-ctdna-methylation",
      "slug": "plasma-ctdna-methylation",
      "title": "Select a plasma ctDNA methylation detection workflow",
      "question": "Which measured methylation workflow detects tumour-derived plasma DNA and, where separately supported, identifies tissue of origin?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "plasma ctdna methylation"
      ],
      "intended_users": [
        "Clinical researchers scoping the available diagnostic-genomics evidence",
        "Computational researchers comparing exact evaluated configurations"
      ],
      "decision": "Inspect the cfMethyl-Seq repeated-split sensitivity/specificity evidence before selecting a workflow and validation design for the intended cancer-detection population.",
      "inputs": [
        "Plasma cell-free DNA for targeted/genome-wide methylation sequencing",
        "The target specificity constraint for the intended use"
      ],
      "output": "A sourced all-stage cancer-detection sensitivity/specificity figure from a repeated random test split; no tissue-of-origin claim is ingested in this bounded intake.",
      "setting": "cfMethyl-Seq achieves 80.7% all-stage cancer-detection sensitivity (95% CI 68.6-90.7) at 97.9% specificity, under a repeated random 25% test split (cohort 217 cancers/191 non-cancers, test n=102).",
      "exclusions": [
        "Tissue-of-origin identification (not ingested in this bounded intake)",
        "Prospective screening validation"
      ],
      "clinical_scope": "Clinical applicability is not established as prospective screening performance: this is a repeated random test-split evaluation on one assembled cohort, not an independent prospective validation.",
      "evidence_gaps": [
        "No per-run scored denominator beyond the printed test n=102 is inferred.",
        "A correction/source-reuse note on this source is tracked separately from the assay sensitivity/specificity values and must not be conflated with them."
      ],
      "citations": [
        {
          "source_id": "amp-source-cfmethyl",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        },
        {
          "source_id": "amp-source-cfmethyl-correction",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Which measured methylation workflow detects tumour-derived plasma DNA and, where separately supported, identifies tissue of origin?",
        "baselines": [
          "The conventional/author-introduced workflow measured in the linked primary source(s)."
        ],
        "outcomes": [
          "The declared endpoint in this use case's active mapping(s); see evidence_gaps for what remains open."
        ],
        "validation_requirements": [
          "Independent held-out population matched to the intended clinical setting.",
          "Qualified human scientific review before any clinical-validation claim."
        ],
        "next_step": "A dedicated rewire-benchmarks protocol/run task with an independent prospective cohort and an explicit tissue-of-origin evaluation if pursued."
      },
      "planned_work": [
        {
          "title": "Focused benchmark/run task for rewire-benchmark-data issue #16",
          "url": "https://github.com/rewire-bio/rewire-benchmark-data/issues/16",
          "status": "planned",
          "reason": "Reviewed literature evidence is bounded (see evidence_gaps); closing the remaining decision gap needs a dedicated protocol/run task with frozen population and matched controls, tracked in rewire-benchmarks."
        }
      ]
    },
    {
      "id": "use-case-protein-stability",
      "slug": "protein-variant-stability",
      "title": "Assess methods for protein stability experiments",
      "question": "What evidence supports ranking protein substitutions by folding stability, and what must be validated before choosing a method?",
      "area": "proteins-complexes",
      "contexts": [
        "research",
        "clinical_research"
      ],
      "search_terms": [
        "protein variant",
        "amino acid substitution",
        "folding stability",
        "protein engineering",
        "proteolysis",
        "AMFR",
        "ProteinGym",
        "ESM-2",
        "EVmutation",
        "EVCouplings",
        "pathogenicity"
      ],
      "intended_users": [
        "Experimental researchers selecting variants for protein stability experiments",
        "Computational researchers evaluating sequence-based variant rankings",
        "Translational researchers checking whether stability evidence answers a disease question"
      ],
      "decision": "Inspect the available AMFR assay results as a narrow example, then identify the missing singles-only comparison and validation required for the target protein and endpoint.",
      "inputs": [
        "Wild-type protein sequence and amino acid substitutions",
        "A defined construct and experimental stability endpoint; alignment-derived methods additionally require a traceable alignment and model"
      ],
      "output": "Exact existing AMFR configurations, their assay scope and unresolved comparisons; no universal model ranking or pathogenicity prediction.",
      "setting": "The current evidence is limited to the 47-residue AMFR_HUMAN_Tsuboyama_2023_4G3O construct in ProteinGym v1.3. Its cDNA-display proteolysis assay infers folding stability. Completed evaluations cover 2,972 variants: 820 single and 2,152 double substitutions.",
      "exclusions": [
        "Whole-protein function, cellular activity, organismal fitness and clinical pathogenicity",
        "Generalisation to other proteins, other ESM-2 checkpoints or the full ProteinGym track",
        "Treating the existing mixed cohort as a completed matched single-substitution comparison"
      ],
      "clinical_scope": "Clinical applicability is not established. Stability of this short experimental construct is not evidence of clinical pathogenicity or suitability for diagnosis or treatment.",
      "evidence_gaps": [
        "The existing ESM-2 and fixed-seed random results use separate protocols. They are displayed separately and do not establish a matched cross-protocol winner.",
        "No uncertainty intervals or seed-variability estimates are recorded for these completed evaluations. One random ranking is not a chance-performance interval.",
        "The proposed ESM-2 versus EVCouplings site-independent and EVmutation comparison targets a frozen set of covered single substitutions. It has no completed results.",
        "The planned comparison still needs a separate execution decision, a traceable real evolutionary model, the full-model adapter, resource logging, frozen populations and tested analysis code. Planning resource ceilings are not measured requirements.",
        "The recorded ESM-2 timer excludes checkpoint loading, preparation and metrics; peak memory is unreported. It is not an end-to-end or cross-model speed comparison.",
        "The existing AMFR protocols have no reviewed direct task-membership link. The mappings therefore reference the protocols only and do not infer membership from the ProteinGym suite.",
        "Assay bytes were hashed locally without independent authentication against an upstream published checksum. Training overlap, independent reproduction and human scientific review remain unresolved.",
        "30 September 2026 audit: The official comparison contains 820 single and 2,152 double substitutions. It does not complete the planned matched singles-only local study.",
        "30 September 2026 audit: Number of Mutants=2,972 is the assay count; exact scored count per model, source score transformations and model checkpoint hashes remain unextracted.",
        "30 September 2026 audit: Method input modalities differ. These scores support short-construct folding-stability ranking, not a universal sequence-only winner, full-protein function or clinical pathogenicity."
      ],
      "citations": [
        {
          "source_id": "profile-protocol-proteingym-reference-files-dms-substitutions-csv-a8f49801",
          "locator": "DMS_id=AMFR_HUMAN_Tsuboyama_2023_4G3O: seq_len, selection_assay, selection_type, raw_DMS_phenotype_name, DMS_number_single_mutants, DMS_number_multiple_mutants"
        },
        {
          "source_id": "rewire-local-20260920-source-proteingym-esm2",
          "locator": "/coverage; /execution; /provenance; /protocol_results"
        },
        {
          "source_id": "rewire-local-20260921-source-proteingym-random",
          "locator": "/coverage; /model_configuration; /protocol_results"
        },
        {
          "source_id": "use-case-source-amfr-pilot-plan-194a78b",
          "locator": "Status; Question; Task and data; Methods; Gates and execution order; Runtime and cost; Roles and review"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "planned_work": [
        {
          "title": "Matched single-substitution comparison: ESM-2, site-independent model and EVmutation",
          "url": "https://benchmarks.rewire.it/omics/sources/a3208aa4537e8d28c56f68f4559299f31241c50b8d8b2b3776cba5f1718caa43.md",
          "status": "blocked",
          "reason": "Planning is complete; execution is not authorised by that plan. Gates require an execution decision, traceable evolutionary model, full-model adapter, resource logging, frozen populations and tested analysis. No measured comparison or winner is available."
        }
      ]
    },
    {
      "id": "use-case-rare-disease-candidate-ranking",
      "slug": "rare-disease-candidate-ranking",
      "title": "Rank rare-disease variants for review",
      "question": "Which methods recover causal variants and genes within a realistic laboratory review budget?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "rare disease",
        "variant prioritisation",
        "gene ranking",
        "exome sequencing",
        "genome sequencing",
        "phenotype",
        "pedigree",
        "Exomiser"
      ],
      "intended_users": [
        "Diagnostic scientists",
        "Clinical genomicists evaluating interpretation workflows"
      ],
      "decision": "Choose methods for prioritising variants in an initial exome or genome analysis, using phenotype, pedigree and molecular evidence to focus laboratory review.",
      "inputs": [
        "Quality-controlled variant calls, genome build and calling coverage",
        "Phenotype terms, pedigree, inheritance and available family sequencing",
        "Dated population-frequency and gene–disease evidence"
      ],
      "output": "A ranked candidate list with supporting evidence, conflicts, filter reasons and unresolved findings for professional review.",
      "setting": "Initial ES/GS interpretation in a defined congenital-anomaly or developmental-disorder population. Singleton and family-based analysis require separate comparisons.",
      "exclusions": [
        "Variant ranking alone does not establish pathogenicity or a patient diagnosis.",
        "Balanced pathogenic/benign variant classification does not measure case-level diagnostic performance.",
        "Reanalysis, cancer-risk estimation and treatment selection are separate questions."
      ],
      "clinical_scope": "Clinical research on interpretation support. Confirmed diagnoses and downstream management outcomes require separate evaluation from candidate retrieval.",
      "evidence_gaps": [
        "Known callable diagnosis recovery is available, but no controlled addition of a molecular-effect model at identical review effort was established.",
        "Family, site and temporal independence, calling failures and unresolved outcomes need explicit prospective evaluation.",
        "Talos/Exomiser comparison has source denominator discrepancies (194 body versus 190 extended figure); rank limits are not equal effort.",
        "Qualified human scientific review remains outstanding; automated source transcription does not establish experimental replication."
      ],
      "citations": [
        {
          "source_id": "use-case-source-clinical-priorities-2026-09-28",
          "locator": "C1 — Rare-disease candidate ranking"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Does adding a molecular-effect model to phenotype-, pedigree- and frequency-aware analysis improve causal-finding recovery at the same analyst review budget?",
        "baselines": [
          "Conventional filtering and phenotype/pedigree-aware review with pinned versions",
          "An appropriate pinned Exomiser configuration without the added model"
        ],
        "outcomes": [
          "Case-level recovery of independently adjudicated causal findings within the review budget",
          "Analyst effort, missed variant classes, unresolved cases and abstention",
          "Confirmed diagnoses measured separately from ranked candidates"
        ],
        "validation_requirements": [
          "Separate families, centres and evaluation time periods.",
          "Freeze knowledge releases and model training cutoffs; audit predictor-derived labels.",
          "Retain unresolved cases in workload and coverage denominators.",
          "Record calling failures separately and include them in end-to-end sensitivity.",
          "Use independent qualified clinical-genetics adjudication for diagnostic claims."
        ],
        "next_step": "Close the remaining endpoint, independence and comparator gaps documented in the 30 September 2026 audit and linked article issue; obtain qualified scientific review."
      },
      "planned_work": []
    },
    {
      "id": "use-case-regulatory-variant-gene-follow-up",
      "slug": "regulatory-variant-gene-follow-up",
      "title": "Select regulatory variants and genes for functional follow-up",
      "question": "Which variants, regulatory elements and genes should I perturb to explain a disease-associated locus?",
      "area": "dna-genomes",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "regulatory variant",
        "effector gene",
        "variant-to-gene",
        "enhancer-to-gene",
        "GWAS",
        "fine mapping",
        "CRISPRi",
        "MPRA",
        "endogenous editing",
        "AlphaGenome",
        "ENCODE",
        "scE2G"
      ],
      "intended_users": [
        "Functional geneticists",
        "Researchers interpreting disease-associated loci"
      ],
      "decision": "Choose allele edits, regulatory-element perturbations and gene readouts that can distinguish competing explanations for a locus.",
      "inputs": [
        "Fine-mapped variants with genome build, ancestry and linkage-disequilibrium context",
        "Candidate regulatory elements, genes and a relevant cell type",
        "Sequence and available chromatin, expression or contact measurements",
        "The feasible perturbation assay, readout and follow-up budget"
      ],
      "output": "A prioritised set of variant–element–gene hypotheses, with the experiments needed to test each link and effect direction where supported.",
      "setting": "Research follow-up of defined disease-associated loci in a specified cell system. Allele-effect prediction, element–gene linking and experimental shortlist selection are separate comparisons.",
      "exclusions": [
        "Reporter activity does not by itself establish endogenous regulation.",
        "Perturbing an entire regulatory element is different from editing one allele.",
        "A regulatory link alone does not establish a gene’s causal role in disease."
      ],
      "clinical_scope": "Research only. Molecular effects and regulatory links do not establish clinical variant classification or diagnostic validity.",
      "evidence_gaps": [
        "Automated source review only; independent human scientific review remains outstanding.",
        "No allele-specific endogenous-edit benchmark or equal-budget prospective shortlist benefit established.",
        "Training/test independence unresolved: paper explicitly says benchmarking pipeline does not perform cross-validation."
      ],
      "citations": [
        {
          "source_id": "use-case-source-research-priorities-2026-09-28",
          "locator": "R2 — Regulatory variant and effector-gene follow-up"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Does a method select more experimentally supported regulatory hypotheses than compatible distance or linking baselines at the same follow-up budget?",
        "baselines": [
          "Distance-based variant/gene or enhancer/gene selection where applicable",
          "ABC, ENCODE-rE2G or scE2G for compatible element–gene linking tasks",
          "Sequence predictors compared only on the allele-effect outputs they produce"
        ],
        "outcomes": [
          "Allele-dependent regulatory effects",
          "Endogenous element–gene and allele–gene effects, measured separately",
          "Confirmed follow-up hits per assay budget"
        ],
        "validation_requirements": [
          "Define the allele-effect, linking or selection endpoint before choosing a comparison.",
          "Hold out loci and studies; audit linkage-disequilibrium and pretraining overlap.",
          "Preserve cell context, genome build, assay and perturbation type.",
          "Keep reporter activity, whole-element perturbation, single-base effects and disease causality distinct."
        ],
        "next_step": "Close the remaining endpoint, independence and comparator gaps documented in the 30 September 2026 audit and linked article issue; obtain qualified scientific review."
      },
      "planned_work": []
    },
    {
      "id": "use-case-rhodopsin-wavelength-transfer",
      "slug": "rhodopsin-wavelength-transfer",
      "title": "Assess rhodopsin wavelength prediction across sequence backgrounds",
      "question": "Which sequence-based configurations merit testing before selecting rhodopsins with different absorption wavelengths in a new wild-type background?",
      "area": "proteins-complexes",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "rhodopsin",
        "opsin",
        "spectral tuning",
        "absorption wavelength",
        "optogenetics",
        "protein engineering",
        "FLIP2",
        "Rhomax",
        "wild-type transfer",
        "ESM-2",
        "sequence composition",
        "ridge regression"
      ],
      "intended_users": [
        "Protein engineers planning rhodopsin spectral-tuning experiments",
        "Computational researchers checking transfer to new sequence backgrounds"
      ],
      "decision": "Use the existing held-out-background evaluations to choose controls and exact model configurations for a new, prospectively held-out spectral-tuning experiment.",
      "inputs": [
        "Complete rhodopsin amino-acid sequences",
        "Training measurements of peak absorption wavelength and frozen assignments that separate wild-type backgrounds"
      ],
      "output": "Evidence for the exact frozen sequence probes and controls, plus the validation still needed for a new background; no calibrated wavelength or wet-lab success guarantee.",
      "setting": "One complete FLIP2 Rhomax by_wild_type test split: 584 training, 116 unused validation and 184 test records. The experimental endpoint is peak absorption wavelength in nanometres.",
      "exclusions": [
        "Rhodopsin activation efficiency, expression, photostability and cellular function",
        "Zero-shot likelihood scoring, alternative pooling or fine-tuning not evaluated in these runs",
        "General protein fitness, all FLIP2 landscapes and clinical use"
      ],
      "clinical_scope": "Research only. Absorption-wavelength ranking does not establish suitability for an optogenetic intervention, patient care or treatment.",
      "evidence_gaps": [
        "A spectral-tuning endpoint is not opsin activation efficiency, expression, photostability, cellular function, general protein fitness or clinical usefulness.",
        "Spearman and full-ranking NDCG do not establish wavelength calibration, top-k precision or a prospective experimental hit rate. Constant predictions have undefined Spearman; their NDCG is a control value, not strong predictive evidence.",
        "No intervals or seed-variability estimates. The 35M checkpoint was chosen after seeing the 8M outcome; the source describes an exploratory follow-up, not a preregistered family-scale comparison.",
        "ESM-2 pretraining overlap is unresolved. A frozen linear probe result does not establish the value of alternative pooling, fine-tuning or every model-family configuration.",
        "Recorded timing covers sections of execution and is not a hardware-normalised deployment-cost comparison. Human domain review and external replication are not established.",
        "30 September 2026 audit: Original RhoMax splits group 75 WT backgrounds (65 train/10 test per split) and are not the single FLIP2 by_wild_type partition. Do not pool their scores.",
        "30 September 2026 audit: Reported standard deviations describe error dispersion; printed aggregate is a summary of four splits, not an independent experiment or pooled uncertainty interval.",
        "30 September 2026 audit: New-background prospective calibration, target-specific candidate hit rates, expression, activation and photostability remain unestablished.",
        "30 September 2026 audit: RhoMax Tables2–3 are feature/attention ablations; they were inspected but intentionally not added to this model-selection comparison."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260921-instructions-esm2-35m",
          "locator": "Matched Rhomax evaluations; Procedure and evidence; Re-run or inspect"
        },
        {
          "source_id": "rewire-local-20260921-source-esm2-35m",
          "locator": "/metrics; /coverage; /model_configuration; /protocol_configuration; /execution; /provenance"
        },
        {
          "source_id": "rewire-local-20260921-source-esm2-8m",
          "locator": "/metrics; /coverage; /model_configuration; /protocol_configuration; /execution; /provenance"
        },
        {
          "source_id": "rewire-local-20260921-source-composition22",
          "locator": "/metrics; /model_configuration; /protocol_configuration"
        },
        {
          "source_id": "rewire-local-20260920-source-flip2-train-mean",
          "locator": "/metrics; /coverage; /model_configuration"
        },
        {
          "source_id": "rewire-local-20260920-source-flip2-composition",
          "locator": "/metrics; /coverage; /model_configuration; /provenance"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "planned_work": []
    },
    {
      "id": "use-case-somatic-small-variant-oncogenicity",
      "slug": "somatic-small-variant-oncogenicity",
      "title": "Assess somatic small-variant oncogenicity",
      "question": "Which methods help classify somatic SNVs and small indels while preserving uncertainty and the relevant gene mechanism?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "somatic variant",
        "oncogenicity",
        "SNV",
        "small indel",
        "oncogene",
        "tumour suppressor",
        "ClinGen",
        "CGC",
        "VICC"
      ],
      "intended_users": [
        "Somatic variant curators",
        "Molecular pathologists evaluating interpretation methods"
      ],
      "decision": "Choose methods that support evidence-based oncogenicity review of somatic small variants and identify when the available evidence remains insufficient.",
      "inputs": [
        "Variant, genome build, transcript, allele fraction, quality and confidence in somatic origin",
        "Tumour context, gene mechanism and dated oncogenicity-specific assertions",
        "Functional studies and the applicable criteria specification"
      ],
      "output": "An oncogenicity classification or unresolved assessment with constituent criteria, conflicting findings and evidence needs.",
      "setting": "SNVs and small indels, with separate protocols for oncogene missense and tumour-suppressor loss-of-function mechanisms.",
      "exclusions": [
        "Inherited predisposition, fusions, rearrangements and copy-number variants require separate interpretation protocols.",
        "Oncogenicity and molecular activity do not establish drug response or clinical actionability.",
        "Germline-benign labels are not automatically valid somatic negative controls."
      ],
      "clinical_scope": "Clinical research on somatic interpretation support. Tumour-only sequencing does not establish somatic origin, and therapeutic relevance needs a separate context-specific review.",
      "evidence_gaps": [
        "OncoVI direct SOP/ClinVar classification exists; prospective reviewer-effort, calibration and independent model-versus-expert evaluation remain missing.",
        "MTB old labels measure protein-function impact, not oncogenicity; selected reassessment shares experts/resources and is disagreement-enriched.",
        "Abstract/body accuracy conflicts remain flagged. Supplemental Table S6 download returned challenge HTML; only body-printed strata are extracted.",
        "Qualified human scientific review remains outstanding; automated source transcription does not establish experimental replication."
      ],
      "citations": [
        {
          "source_id": "use-case-source-clinical-priorities-2026-09-28",
          "locator": "C4 — Somatic small-variant oncogenicity"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Does a proposed method improve independent oncogenicity classification or review efficiency over explicit conventional criteria while retaining calibrated unresolved results?",
        "baselines": [
          "ClinGen/CGC/VICC criteria-based review under a pinned specification",
          "A relevant specialised predictor assessed only within its actual output scope"
        ],
        "outcomes": [
          "Category precision/recall and criteria fidelity",
          "Calibration, abstention and unresolved-case coverage",
          "Expert review effort, measured separately from treatment outcomes"
        ],
        "validation_requirements": [
          "Use oncogenicity-specific labels with evidence, review status and dates.",
          "Hold out allelic series, residues, studies and genes where unseen-gene transfer is intended.",
          "Audit training and assertion overlap, including predictor contributions to labels.",
          "Preserve tumour, assay and gene-mechanism context.",
          "Use independent qualified somatic-variant adjudication and justified negative or uncertain examples."
        ],
        "next_step": "Close the remaining endpoint, independence and comparator gaps documented in the 30 September 2026 audit and linked article issue; obtain qualified scientific review."
      },
      "planned_work": []
    },
    {
      "id": "use-case-splicing-follow-up",
      "slug": "splicing-follow-up",
      "title": "Prioritise variants for splicing experiments",
      "question": "Which evaluated configurations can inform selection of human SNVs for follow-up splicing experiments?",
      "area": "dna-genomes",
      "contexts": [
        "research",
        "clinical_research"
      ],
      "search_terms": [
        "splicing",
        "splice disruption",
        "SNV",
        "single nucleotide variant",
        "variant prioritisation",
        "exon recognition",
        "minigene",
        "patient RNA",
        "MFASS",
        "SpliceAI",
        "Pangolin"
      ],
      "intended_users": [
        "Experimental researchers selecting variants for functional follow-up",
        "Computational researchers comparing splice-effect configurations",
        "Clinical researchers investigating the limits of assay evidence"
      ],
      "decision": "Inspect the matched MFASS configurations and their limitations before selecting a method and designing validation in the intended experimental setting.",
      "inputs": [
        "Human single nucleotide variants with alleles, genome assembly and gene/transcript context",
        "The intended experimental endpoint and the number of variants that can be followed up"
      ],
      "output": "A sourced set of exact evaluated configurations and remaining validation needs; no patient-level variant classification or recommendation.",
      "setting": "MFASS measures exon recognition in an artificial minigene reporter. The matched study evaluates genomic-context SpliceAI and Pangolin configurations against this functional endpoint on a fixed held-out population.",
      "exclusions": [
        "Indels and variant types outside the reviewed human SNV protocol",
        "Patient-RNA effects, disease pathogenicity, diagnostic yield and treatment decisions",
        "Pooling the matched annotation study with historical MFASS runs using other annotations or scoring populations"
      ],
      "clinical_scope": "Clinical applicability is not established. Reporter-assay ranking does not demonstrate patient-RNA performance, pathogenicity classification or clinical yield; validation in the intended population and workflow is still needed.",
      "evidence_gaps": [
        "All four conditions scored 8,297 of 8,324 held-out variants. The same 27 exclusions comprise 23 hg19-to-hg38 assembly-orientation mismatches and four canonical-transcript-span exclusions; the latter are not established faulty variants. Missing predictions are not negative predictions.",
        "No top-100 precision difference is established; Pangolin's masked top-100 result is sensitive to the registered tie order. Precision at 100 does not transfer automatically to another follow-up capacity or prevalence.",
        "Individual-condition uncertainty intervals are not recorded. Paired-contrast intervals concern differences between conditions and must not be shown as each condition's uncertainty.",
        "The study is exploratory: prior outcomes were inspected and nine contrast intervals are unadjusted. Matching annotation does not isolate model architecture.",
        "Author confirmation of the assembly-orientation finding is not established. Human scientific review and independent replication remain outstanding.",
        "These mappings do not change the MFASS task's discovered status. A source-reviewed applicability mapping is separate from reviewing the task record.",
        "30 September 2026 audit: The four matched conditions exclude the same 27 of 8,324 held-out variants; missing scores are not negative predictions.",
        "30 September 2026 audit: Historical v2 marginal scores have different scored subsets/annotations and require their own paired comparisons. Constant-prior top-100 counts reflect tied-score order. DNABERT2 is a pipeline rather than an exact configuration and remains outside the active mapping.",
        "30 September 2026 audit: MFASS reporter exon inclusion does not validate patient RNA, diagnosis or prospective clinical follow-up. Human scientific review and external reproduction remain outstanding."
      ],
      "citations": [
        {
          "source_id": "evidence-expansion-mfass-readme-62a93814",
          "locator": "Dataset; Limits: artificial minigene exon-recognition endpoint, not patient RNA"
        },
        {
          "source_id": "use-case-source-mfass-matched-intake-194a78b",
          "locator": "Opening paragraphs: coverage, exclusion classes, annotation, review and interpretation; Review and validation"
        },
        {
          "source_id": "rewire-mfass-matched-v1-source-report",
          "locator": "/conditions/{S0,S1,P0,P1}/coverage; /conditions/{S0,S1,P0,P1}/ties; /contrasts/{S1-S0,P1-P0,P0-S0}/paired/precision_at_capacity"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "planned_work": []
    },
    {
      "id": "use-case-structural-hypotheses-experiments",
      "slug": "structural-hypotheses-experiments",
      "title": "Choose structural hypotheses to guide experiments",
      "question": "Which predicted interfaces or structures are reliable enough to guide my next experiment?",
      "area": "proteins-complexes",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "structural hypothesis",
        "protein complex",
        "interface mutation",
        "construct selection",
        "protein structure",
        "experimental planning",
        "AlphaFold",
        "FoldBench",
        "CASP",
        "CAPRI",
        "confidence calibration"
      ],
      "intended_users": [
        "Structural biologists",
        "Researchers selecting interface mutations or protein constructs"
      ],
      "decision": "Choose structural hypotheses and associated mutations or constructs for experimental testing.",
      "inputs": [
        "Protein sequences, partner identities and the intended assembly",
        "Available experimental structures or templates with release dates",
        "A defined experimental choice and testing budget",
        "Relevant biochemical conditions and alternative conformations"
      ],
      "output": "Structural hypotheses with confidence and coverage estimates, linked to proposed mutations or constructs and tests of their reliability.",
      "setting": "Research planning within a bounded class of protein complexes. Monomers, antibody complexes and ligand complexes require separate comparison protocols.",
      "exclusions": [
        "Structural accuracy alone does not establish that a proposed experiment will succeed.",
        "High confidence does not prove interaction existence, affinity or function.",
        "Results from one complex class do not establish performance in another."
      ],
      "clinical_scope": "Research only. Structural hypotheses do not establish therapeutic efficacy or clinical suitability.",
      "evidence_gaps": [
        "Automated source review only; independent human scientific review remains outstanding.",
        "No intervals in Table 3; no prospective experimental utility or matched conventional mutation/construct baseline.",
        "Scored target populations differ; Table 1 assessable counts are retained as coverage context, not verified metric denominators. Some Table 3 rates do not reconcile with integer counts after rounding. No complete-cohort estimate is inferred.",
        "ClusPro uses component 3D structures and BM5 targets, so scores cannot be directly compared with FoldBench; BM4 parameter benchmarking limits holdout independence. Exact ClusPro revision and input hashes remain unextracted.",
        "ClusPro Total/easy/top10 source prints 87 although subtotals and 51.68% imply 77; literal 87 is retained as needs_review outside use-case mappings. Antibody and aggregate classes are not mapped to this issue."
      ],
      "citations": [
        {
          "source_id": "use-case-source-research-priorities-2026-09-28",
          "locator": "R5 — Structural hypotheses for experiments"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Do model-derived structural hypotheses improve experimental choices over appropriate conventional strategies at the same testing budget?",
        "baselines": [
          "Suitable template modelling or docking for the specified complex class",
          "Conventional mutation or construct selection for experimental-utility comparisons"
        ],
        "outcomes": [
          "Task-specific contact or pose accuracy and prediction failures as intermediate evidence",
          "Confidence calibration and coverage",
          "Independent interface-mutation or construct success where measured"
        ],
        "validation_requirements": [
          "Define the experimental choice and compatible complex class before selecting metrics.",
          "Audit sequence, template, ligand and structure-deposition overlap with training data.",
          "Include failed predictions and relevant alternative conformations.",
          "Keep structural accuracy, interaction existence, affinity, function and experimental utility distinct."
        ],
        "next_step": "Close the remaining endpoint, independence and comparator gaps documented in the 30 September 2026 audit and linked article issue; obtain qualified scientific review."
      },
      "planned_work": []
    },
    {
      "id": "use-case-therapeutic-target-validation",
      "slug": "therapeutic-target-validation",
      "title": "Select therapeutic targets for validation",
      "question": "Which targets should I test to change a defined disease-relevant phenotype?",
      "area": "cells-tissues",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "target prioritisation",
        "target validation",
        "therapeutic target",
        "disease mechanism",
        "direction of effect",
        "genetic dependency",
        "Open Targets",
        "DepMap"
      ],
      "intended_users": [
        "Disease biologists",
        "Translational discovery teams"
      ],
      "decision": "Select targets and modulation directions for a validation batch, with clear tests of the proposed disease mechanism.",
      "inputs": [
        "A defined disease, biological context and experimental phenotype",
        "A candidate gene set and feasible ways to inhibit or activate each target",
        "Dated genetic, expression and functional evidence, including conflicting findings",
        "A validation budget and relevant normal-cell or selectivity controls"
      ],
      "output": "A shortlist of target–disease–intervention hypotheses with supporting evidence, uncertainties and proposed validation experiments.",
      "setting": "Research planning for one disease and a prespecified experimental system. The intended comparison tests whether rankings improve the yield of useful target effects.",
      "exclusions": [
        "Untested targets cannot be treated as negative examples.",
        "Cellular dependency, target tractability and a therapeutic window require different evidence.",
        "A target association does not establish that modulating it will benefit patients."
      ],
      "clinical_scope": "Research only. Target prioritisation does not establish treatment benefit or support patient-specific treatment decisions.",
      "evidence_gaps": [
        "61 nominations / Results 57 selected targets / README 59 targets / 50 post-QC perturbations need reconciliation; no success fraction calculated.",
        "Automated source review only; independent human scientific review remains outstanding.",
        "Custom ranking AUC is not ROC AUC, prospective hit rate or efficacy.",
        "No matched-budget conventional/random prospective comparison; no antitumor efficacy, rescue or selectivity result in this endpoint.",
        "Original model implementations/checkpoints are not pinned in the intake; no uncertainty printed.",
        "Preprint v1; original and reimplemented ranking scores kept separate. Screen2 ranking population is enriched by original nomination methods."
      ],
      "citations": [
        {
          "source_id": "use-case-source-research-priorities-2026-09-28",
          "locator": "R1 — Therapeutic target validation"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Does a proposed ranking identify more reproducible disease-relevant target effects than conventional evidence aggregation at the same validation budget?",
        "baselines": [
          "Genetics-only and expression-only rankings",
          "Simple dependency ranking or conventional evidence aggregation"
        ],
        "outcomes": [
          "Confirmed useful target effects among all candidates tested",
          "Reproducibility, rescue and context/selectivity",
          "Failed experiments and uncertainty about modulation direction"
        ],
        "validation_requirements": [
          "Define the disease, candidate universe, intervention direction, phenotype and budget before comparing rankings.",
          "Freeze input evidence before validation outcomes and audit overlap between discovery and validation data.",
          "Include failed experiments and normal-cell controls; retain untested candidates as untested.",
          "Keep target association, dependency, tractability and clinical success separate."
        ],
        "next_step": "Close the remaining endpoint, independence and comparator gaps documented in the 30 September 2026 audit and linked article issue; obtain qualified scientific review."
      },
      "planned_work": []
    },
    {
      "id": "use-case-tumour-dna-somatic-variant-detection",
      "slug": "tumour-dna-somatic-variant-detection",
      "title": "Select a tumour DNA somatic variant-calling workflow",
      "question": "Which calling or rescoring workflow reliably detects tumour SNVs and small indels under the available sequencing and normal-sample regime?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "tumour dna somatic variant detection"
      ],
      "intended_users": [
        "Clinical researchers scoping the available diagnostic-genomics evidence",
        "Computational researchers comparing exact evaluated configurations"
      ],
      "decision": "Inspect the matched virtual-tumour precision/recall evidence for Lancet and Strelka2 before selecting a caller and designing a validation protocol for the intended sequencing/normal regime.",
      "inputs": [
        "Tumour and matched-normal aligned sequencing reads",
        "The intended variant-fraction, coverage and genomic-context strata"
      ],
      "output": "A sourced set of exact evaluated caller precision/recall/F1 values and remaining validation needs; no patient-level variant classification or clinical recommendation.",
      "setting": "A real-read virtual-tumour spike-in (NA12892/NA12891 HapMap samples, 80x/40x WGS) measures SNV and indel precision/recall/F1 for the Lancet caller and the Strelka2 comparator.",
      "exclusions": [
        "Real clinical tumour specimens, FFPE samples, targeted panels, ctDNA, copy-number and structural variation",
        "Callers other than Lancet/Strelka2 not transcribed in this bounded intake (Strelka, MuTect, MuTect2, LoFreq rows exist in the source but are not all ingested; LoFreq/MuTect2 unselected rows are internally inconsistent in the source and are excluded)"
      ],
      "clinical_scope": "Clinical applicability is not established. A synthetic virtual-tumour spike-in does not demonstrate sensitivity in real clinical tumours, FFPE specimens, gene panels, ctDNA, or for copy-number/structural variants.",
      "evidence_gaps": [
        "Individual-condition confidence intervals are unreported in the source tables.",
        "Exact caller release versions are not extracted from the primary article.",
        "No foundation-model component is evaluated in this bounded intake."
      ],
      "citations": [
        {
          "source_id": "amp-oncology-rna-20261007-source-pmc6123722",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Which calling or rescoring workflow reliably detects tumour SNVs and small indels under the available sequencing and normal-sample regime?",
        "baselines": [
          "The conventional/author-introduced workflow measured in the linked primary source(s)."
        ],
        "outcomes": [
          "The declared endpoint in this use case's active mapping(s); see evidence_gaps for what remains open."
        ],
        "validation_requirements": [
          "Independent held-out population matched to the intended clinical setting.",
          "Qualified human scientific review before any clinical-validation claim."
        ],
        "next_step": "A dedicated rewire-benchmarks protocol/run task with a real clinical tumour cohort, pinned truth set, purity/platform/assembly strata and an explicit foundation-model eligibility check."
      },
      "planned_work": [
        {
          "title": "Focused benchmark/run task for rewire-benchmark-data issue #10",
          "url": "https://github.com/rewire-bio/rewire-benchmark-data/issues/10",
          "status": "planned",
          "reason": "Reviewed literature evidence is bounded (see evidence_gaps); closing the remaining decision gap needs a dedicated protocol/run task with frozen population and matched controls, tracked in rewire-benchmarks."
        }
      ]
    },
    {
      "id": "use-case-tumour-rna-fusion-detection",
      "slug": "tumour-rna-fusion-detection",
      "title": "Select a tumour RNA fusion-detection workflow",
      "question": "Which RNA-sequencing workflow detects and prioritises tumour gene fusions with useful sensitivity and a manageable false-positive review burden?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "tumour rna fusion detection"
      ],
      "intended_users": [
        "Clinical researchers scoping the available diagnostic-genomics evidence",
        "Computational researchers comparing exact evaluated configurations"
      ],
      "decision": "Inspect the matched synthetic Seraseq reference-standard and NCH clinical-ascertainment evidence for Arriba, STAR-Fusion and the EnFusion ensemble before selecting a workflow and review burden for the intended specimen type.",
      "inputs": [
        "Paired-end tumour RNA-seq reads",
        "The intended specimen type (frozen, FFPE or other) and acceptable false-positive review burden"
      ],
      "output": "A sourced set of exact evaluated sensitivity/precision values on a synthetic reference standard and a real clinical cohort, with the ascertainment-bias caveat explicit; no patient-level fusion classification.",
      "setting": "A 14-fusion synthetic Seraseq reference standard (undiluted, duplicate libraries) and a 229-sample paediatric cancer/haematologic cohort (67 ensemble-ascertained clinically relevant fusions) measure fusion-calling sensitivity and precision for Arriba, STAR-Fusion and the EnFusion ensemble (with/without filtering and a known-fusion list).",
      "exclusions": [
        "Novel-fusion discovery performance (the optimised known-fusion rescue measures recovery of 14 known synthetic fusions, not novel-fusion discovery)",
        "Adult solid-tumour cohorts outside the transcribed NCH cohort",
        "Non-Seraseq background fusions, all counted false positive in this intake though some may be real endogenous fusions"
      ],
      "clinical_scope": "Clinical applicability is bounded: the NCH clinical-cohort sensitivity figures are ascertained against the optimised EnFusion ensemble's own calls, not an independent exhaustive truth set, so a false-negative rate for fusions the ensemble itself missed cannot be derived from this evidence.",
      "evidence_gaps": [
        "Table 2 caption (v3) vs the optimisation narrative (v2) is an unresolved source version conflict.",
        "STAR-Fusion precision is printed as both 43.6% (Table 2) and 43.8% (paragraph) in the source; both are preserved, neither is silently reconciled.",
        "No confidence interval or dispersion is printed for the Seraseq Table 2 rows."
      ],
      "citations": [
        {
          "source_id": "amp-oncology-rna-20261007-source-pmc8642973",
          "locator": "See data/omics/use-case-coverage-amp-20261007/clinical/sources.md"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Which RNA-sequencing workflow detects and prioritises tumour gene fusions with useful sensitivity and a manageable false-positive review burden?",
        "baselines": [
          "The conventional/author-introduced workflow measured in the linked primary source(s)."
        ],
        "outcomes": [
          "The declared endpoint in this use case's active mapping(s); see evidence_gaps for what remains open."
        ],
        "validation_requirements": [
          "Independent held-out population matched to the intended clinical setting.",
          "Qualified human scientific review before any clinical-validation claim."
        ],
        "next_step": "A dedicated rewire-benchmarks protocol/run task with an independent fusion truth set not ascertained by the same ensemble being evaluated."
      },
      "planned_work": [
        {
          "title": "Focused benchmark/run task for rewire-benchmark-data issue #11",
          "url": "https://github.com/rewire-bio/rewire-benchmark-data/issues/11",
          "status": "planned",
          "reason": "Reviewed literature evidence is bounded (see evidence_gaps); closing the remaining decision gap needs a dedicated protocol/run task with frozen population and matched controls, tracked in rewire-benchmarks."
        }
      ]
    },
    {
      "id": "use-case-unresolved-rare-disease-reanalysis",
      "slug": "unresolved-rare-disease-reanalysis",
      "title": "Reanalyse unresolved rare-disease cases",
      "question": "Which reanalysis methods find new diagnoses or reduce review effort when given the same updated information?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "rare disease",
        "reanalysis",
        "negative exome",
        "unresolved genome",
        "variant reevaluation",
        "Talos",
        "Exomiser",
        "ClinVar"
      ],
      "intended_users": [
        "Clinical genomics services responsible for unresolved cases",
        "Laboratory reanalysis leads"
      ],
      "decision": "Choose a workflow that identifies previously unresolved cases worth reopening and explains the evidence changes that warrant review.",
      "inputs": [
        "Original ES/GS calls, coverage, report, filters and analysis date",
        "Updated phenotype, pedigree and gene–disease or variant evidence",
        "Dated changes to calling and interpretation methods"
      ],
      "output": "A prioritised queue of new or changed findings, their evidence history and the review or confirmation needed to resolve each case.",
      "setting": "A defined cohort remaining unresolved after an earlier ES/GS analysis, followed over a stated interval. Variant reevaluation and whole-case reanalysis are recorded separately.",
      "exclusions": [
        "Reranking known solved cases does not establish new diagnostic yield.",
        "New knowledge, phenotypes or variant calls must not be counted automatically as algorithmic improvement.",
        "A changed database label or new candidate is not a confirmed diagnosis."
      ],
      "clinical_scope": "Clinical research on reanalysis support. Programme-level yield and the added value of a method require different comparisons; diagnoses and diagnosed individuals have separate denominators.",
      "evidence_gaps": [
        "Equal updated calls, phenotypes and knowledge with matched review effort are not established in the extracted peer-reviewed programme.",
        "A 2026 medRxiv automated-versus-manual comparison was discovered (10.64898/2026.05.16.26352295); full text retrieval failed with HTTP403 and indexed percentages conflict between text and caption. No measurements imported from that preprint.",
        "False alerts, review time, retracted diagnoses and source denominator inconsistencies need adjudication.",
        "Qualified human scientific review remains outstanding; automated source transcription does not establish experimental replication."
      ],
      "citations": [
        {
          "source_id": "use-case-source-clinical-priorities-2026-09-28",
          "locator": "C2 — Unresolved rare-disease reanalysis"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "collection_plan": {
        "status": "collecting",
        "comparison_question": "Does a proposed reanalysis method improve confirmed diagnoses or reduce review burden relative to refreshed conventional analysis using identical updated knowledge, phenotypes and calls?",
        "baselines": [
          "Refreshed conventional analysis with the same updated information",
          "The original unresolved cohort for programme-level incremental yield, reported separately"
        ],
        "outcomes": [
          "New confirmed diagnoses per eligible case over a stated interval",
          "Added method benefit under equal updated inputs",
          "False alerts, review time and the reason for each resolved case"
        ],
        "validation_requirements": [
          "Record original and update dates; exclude future knowledge at each analysis time.",
          "Separate benefit from new information, calling changes and prioritisation changes.",
          "Audit family/site overlap and originating-laboratory label circularity.",
          "Count diagnoses and diagnosed individuals separately; distinguish first and repeated cycles.",
          "Use independent qualified clinical-genetics review to confirm outcomes."
        ],
        "next_step": "Close the remaining endpoint, independence and comparator gaps documented in the 30 September 2026 audit and linked article issue; obtain qualified scientific review."
      },
      "planned_work": []
    },
    {
      "id": "use-case-utr-translation-baselines",
      "slug": "utr-translation-baselines",
      "title": "Set baselines for UTR translation experiments",
      "question": "Before training a more complex model of reporter translation, which sequence-only controls should be measured on the same held-out data?",
      "area": "rna-transcriptomes",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "5 prime UTR",
        "5′UTR",
        "translation",
        "mean ribosome load",
        "MRL",
        "reporter assay",
        "synthetic mRNA",
        "mRNABench",
        "Sample",
        "RidgeCV",
        "sequence composition",
        "training mean",
        "baseline"
      ],
      "intended_users": [
        "Computational researchers building models of reporter translation",
        "Experimental researchers evaluating a designed 5′UTR library before selecting a modelling strategy"
      ],
      "decision": "Establish training-mean and sequence-composition controls on a fixed held-out reporter dataset before deciding whether a more complex model adds useful prediction. The current evidence contains these two procedural controls and no pretrained model.",
      "inputs": [
        "Complete processed sequences and measured mean ribosome load from a defined reporter library",
        "A fixed training, validation and test assignment, with test labels withheld from fitting and model selection"
      ],
      "output": "Two exact baseline configurations, their matched held-out evaluations and a repeat recipe; no foundation-model ranking or prediction of therapeutic performance.",
      "setting": "The current comparison covers only the Sample designed subset in mRNABench and its measured mean ribosome load. Both controls use the full processed source sequence and the same seed-2541 split: 70,011 training, 15,003 validation and 15,003 test records. Both score all 15,003 held-out test records.",
      "exclusions": [
        "Benchmark-wide mRNABench performance or comparisons with the paper’s aggregated model scores",
        "Claims about novel combinations of sequence motifs, homology-separated generalisation, other reporter systems or RNA chemistries",
        "Therapeutic potency, in-vivo protein output and clinical decisions"
      ],
      "clinical_scope": "This is research evidence for designing a baseline comparison. Reporter mean ribosome load does not establish therapeutic efficacy or clinical suitability.",
      "evidence_gaps": [
        "Only two procedural controls have been evaluated here; no pretrained model or more complex sequence model has a matched result in this protocol.",
        "There is one split and one execution per configuration, with no uncertainty interval or seed-variability estimate. The split does not establish homology separation or compositional generalisation.",
        "RidgeCV uses training labels only and leaves the validation split unused. This differs from upstream probing; the selected test MSE must not be compared as if it were the paper’s aggregated Pearson score or default validation result.",
        "The training-mean control has undefined Pearson and Spearman correlations because its predictions are constant. Unavailable correlations are not zero performance scores.",
        "The original run’s raw inputs and saved predictions are not publicly archived. The public reports record hashes and a recipe for obtaining source data and running the controls again; prior predictions cannot be rescored from those reports. Dataset reuse terms are unreported.",
        "The recorded repeat recipe executes all four local sequence controls, including a separate protein dataset. Portable command examples have not been rerun verbatim during this review. Timings cover the reported calculation sections, not complete setup or cross-machine performance.",
        "30 September 2026 audit: FramePool/Sample random and truncated-human libraries differ from the locally evaluated mRNABench designed-MRL split. Learned results exist for the use case, but no new result fills that exact local protocol gap.",
        "30 September 2026 audit: Sequence-length transfer is measured; Pearson correlation does not establish absolute calibration, design success, therapeutic translation or full-length endogenous prediction.",
        "30 September 2026 audit: Local protocol raw predictions/input reuse terms remain unresolved; supplementary tables do not establish permissions for those separate inputs.",
        "30 September 2026 audit: Per-configuration uncertainty and exact human cohort denominators remain unextracted."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260920-source-mrnabench-composition",
          "locator": "/coverage; /model_configuration; /protocol_configuration; /protocol_results; /provenance"
        },
        {
          "source_id": "rewire-local-20260920-source-mrnabench-train-mean",
          "locator": "/coverage; /model_configuration; /protocol_results/metric_unavailable_reasons; /provenance"
        },
        {
          "source_id": "rewire-local-20260920-instructions-mrnabench",
          "locator": "Results: RNA evaluation paragraph; Provenance and review; Reproduce the four sequence controls"
        },
        {
          "source_id": "expansion-p3-mrnabench-2025",
          "locator": "Local Tasks (S12); Appendix B.7 Mean Ribosome Load - MPRA (S33); Data Splitting Strategies (S17); Linear Probing (S18); Appendix C (APP3)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "planned_work": []
    }
  ],
  "mappings": [
    {
      "id": "use-case-map-gears-norman-table6-mse",
      "use_case_id": "use-case-genetic-perturbation-response",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "gears-2023-supp-table6-task-mse",
      "evaluation_ids": [
        "gears-2023-supp-table6-evaluation-no-perturb-mse",
        "gears-2023-supp-table6-evaluation-cpa-mse",
        "gears-2023-supp-table6-evaluation-cpa-plus-kg-mse",
        "gears-2023-supp-table6-evaluation-gears-mse"
      ],
      "endpoint": "Mean squared error of predicted versus observed post-perturbation expression; lower is better.",
      "relevance": "proxy",
      "rationale": "The complete comparison includes a no-change control, CPA, CPA with knowledge-graph features and GEARS. It can inform which controls to include in a pilot; aggregate expression agreement does not directly measure follow-up experiment yield.",
      "constraints": [
        "Use the exact Norman2019 K562 source-table protocol and configurations; do not combine this evidence with PerturBench or other GEARS splits.",
        "Read both Table 6 endpoints together while retaining their separate metrics and shared experimental provenance."
      ],
      "limitations": [
        "Exact Table 6 split and scored gene subset remain unextracted; checkpoint and runtime requirements are not established by these catalogue evaluations.",
        "Reported spreads have unspecified type. A numerical difference is not a significance test.",
        "Predictions need validation in the intended cell system before experiment selection."
      ],
      "citations": [
        {
          "source_id": "coverage-source-gears-supp",
          "locator": "Supplementary Table 6, printed page 34 / PDF page 35; Supplementary Table 1, PDF page 30; Supplementary Notes 5 and 14 (K562 context)."
        },
        {
          "source_id": "evidence-official-c037e3419c04936262a0",
          "locator": "README.md at f374e43e197b295016d80395d7a54ddb81cc6769, A note on usage (lines 15–19) and Core API Interface (lines 20–54)."
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "c28653a778ad2953ce30515281cfbcbec052ef644991079b2d503d7ae92c22c2"
    },
    {
      "id": "use-case-map-gears-norman-table6-pearson-de",
      "use_case_id": "use-case-genetic-perturbation-response",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "gears-2023-supp-table6-task-pearson-de",
      "evaluation_ids": [
        "gears-2023-supp-table6-evaluation-no-perturb-pearson-de",
        "gears-2023-supp-table6-evaluation-cpa-pearson-de",
        "gears-2023-supp-table6-evaluation-cpa-plus-kg-pearson-de",
        "gears-2023-supp-table6-evaluation-gears-pearson-de"
      ],
      "endpoint": "Pearson correlation of predicted versus observed expression change from unperturbed controls; higher is better.",
      "relevance": "proxy",
      "rationale": "The complete comparison includes a no-change control, CPA, CPA with knowledge-graph features and GEARS. It can inform which controls to include in a pilot; aggregate expression agreement does not directly measure follow-up experiment yield.",
      "constraints": [
        "Use the exact Norman2019 K562 source-table protocol and configurations; do not combine this evidence with PerturBench or other GEARS splits.",
        "Read both Table 6 endpoints together while retaining their separate metrics and shared experimental provenance."
      ],
      "limitations": [
        "Exact Table 6 split and scored gene subset remain unextracted; checkpoint and runtime requirements are not established by these catalogue evaluations.",
        "Reported spreads have unspecified type. A numerical difference is not a significance test.",
        "Predictions need validation in the intended cell system before experiment selection."
      ],
      "citations": [
        {
          "source_id": "coverage-source-gears-supp",
          "locator": "Supplementary Table 6, printed page 34 / PDF page 35; Supplementary Table 1, PDF page 30; Supplementary Notes 5 and 14 (K562 context)."
        },
        {
          "source_id": "evidence-official-c037e3419c04936262a0",
          "locator": "README.md at f374e43e197b295016d80395d7a54ddb81cc6769, A note on usage (lines 15–19) and Core API Interface (lines 20–54)."
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "87737c4787ced71240c112f2f4cc859131132633b9f029abb0d8aa043f310486"
    },
    {
      "id": "use-case-map-perteval-scfm-norman-single-auspc",
      "use_case_id": "use-case-genetic-perturbation-response",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: a second, independently-sourced, full-text-verified comparison of GEARS (trained from scratch) against conventional baselines (an Xc-plus-co-expression MLP baseline, a context-mean baseline) on a differently-preprocessed Norman2019 single-gene subset, with an independent split mechanism (SPECTRA sparsification) and metric (AUSPC), distinct from the existing GEARS Supplementary Table 6 comparison already mapped to this use case.",
      "protocol_id": "perteval-scfm-2025-protocol-norman-single-2000hvg-auspc",
      "evaluation_ids": [
        "perteval-scfm-2025-evaluation-gears",
        "perteval-scfm-2025-evaluation-mlp-baseline",
        "perteval-scfm-2025-evaluation-mean-baseline"
      ],
      "endpoint": "Area Under the SPECTRA Performance Curve (AUSPC): trapezoidal-rule integral of mean squared error scoring the perturbation-effect delta := P - Xc (Eq. 5, described by the source as a 'log fold change perturbation effect'), across seven increasing-distribution-shift (SPECTRA sparsification) train-test splits; lower is better.",
      "relevance": "proxy",
      "rationale": "A second, independent, full-text-verified expression-change-prediction comparison including GEARS (trained from scratch by a different group of authors), an MLP baseline (input Xc concatenated with perturbed-gene co-expression features, Eq. 3) and a context-mean baseline. It can inform the same pilot-selection decision as the existing Table 6 comparison; it does not measure experiment-selection yield, and it uses a different dataset preprocessing, split mechanism and metric from Table 6, so the two must not be merged.",
      "constraints": [
        "Use the exact PertEval-scFM Table 1 Norman single-gene (2,000 HVG) protocol and configurations; do not combine this evidence with the existing GEARS Supplementary Table 6 comparison, PerturBench, or any other dataset/split/metric in this catalogue.",
        "This intake is limited to the three configurations GEARS, MLP baseline and Mean baseline from Table 1's Norman single-gene section; the five scFM-embedding configurations (Geneformer, scBERT, scFoundation, scGPT, UCE) in the same table, the double-gene section, and the Replogle K562/RPE1 tables in the same paper are not part of this mapping."
      ],
      "limitations": [
        "AUSPC's printed +/- is the source's own propagated standard error (Appendix F.2, Eqs. F3-F5, propagating per-split triplicate-run MSE uncertainty; main-text Figure 2 caption: 'standard error bars') -- a single, author-reported, propagated quantity, not an independently resampled model-seed standard deviation or confidence interval, and its mathematical derivation is not independently validated here.",
        "Per-sparsification-split (S0.1-S0.7) scored perturbation counts are not printed; the paper's Appendix A.1 gives only the eligible raw dataset perturbation count (105 single-gene perturbations), which is not a scored-per-split count and must not be substituted for one.",
        "Scores the perturbation-effect delta := P - Xc (Eq. 5), not raw post-perturbation expression; the same proxy endpoint class as the existing Table 6 mappings (expression-change prediction, not prospective experiment-selection hit rate) but a different metric construction from Table 6's own Pearson-delta-expression. Does not establish anything about case 6's intervention-selection endpoint.",
        "GEARS here is an independent from-scratch training run by this paper's authors (official implementation, only the train-test split modified, other parameters at default values, no pretrained weights), not the original GEARS authors' own Table 6 checkpoint/run; this is a second, independent GEARS evaluation, not a reproduction of Table 6.",
        "The Mean baseline's input population ('all cells in the same context', per the source's own Section 2.2 text) is not fully specified as control-only or training-only; input budgets differ across the three intaken configurations and are not assumed identical.",
        "This protocol concerns generalisation to unseen perturbations under distribution shift; it does not test or establish forecasting for unseen cells, donors, or cell-line/context transfer."
      ],
      "citations": [
        {
          "source_id": "perteval-scfm-2025-source",
          "locator": "Table 1 (Norman single-gene section, rows GEARS / MLP baseline / Mean baseline, AUSPC column, units 10^-2); Appendix A.1 (Norman dataset overview); Section 2.1 (Eq. 3, MLP baseline input); Section 2.2 (Eq. 4 MLP baseline, Eq. 5 perturbation-effect target, GEARS baseline, Mean baseline); Appendix F.2 and Algorithm 1 (AUSPC/uncertainty propagation definitions); main-text Figure 2 caption; Appendix I, Figure I1 caption."
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet genetic-perturbation-response evidence-research worker, independently reviewed by Codex",
        "reviewed_at": "2026-10-07T13:24:02.149Z",
        "note": "Primary-source transcription of three printed table cells (Table 1, Norman single-gene, AUSPC column), independently cross-checked against a freshly re-fetched copy of the same PDF (identical SHA-256), the pinned GitHub code at commit ce48c8b998901c8f8b6275114caac3a4d8543c0b, and the PMLR general publication agreement. No new model execution, independent replication, or qualified human scientific review."
      },
      "evidence_sha256": "3cefc6cec994d3e98ec549910481180ad347e3ee0530c7ba9026e9fc5e8d4f07"
    },
    {
      "id": "use-case-mapping-20260930-335-7acc50652955",
      "use_case_id": "use-case-mass-spectrum-molecule-shortlisting",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-msalign-v2-protocol-massspecgym-mces-formula-free",
      "evaluation_ids": [
        "uc20260930-msalign-v2-eval-massspecgym-mces-formula-free-deepset",
        "uc20260930-msalign-v2-eval-massspecgym-mces-formula-free-jestr",
        "uc20260930-msalign-v2-eval-massspecgym-mces-formula-free-emb-cos",
        "uc20260930-msalign-v2-eval-massspecgym-mces-formula-free-msalign",
        "uc20260930-msalign-v2-eval-massspecgym-mces-formula-free-msalign-score-fusion-overbar"
      ],
      "endpoint": "Formula-free candidate Recall@1, @5 and @20: massspecgym-mces",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "Formula-oracle comparator records excluded from this use-case mapping.",
        "Do not pool formula and MCES splits or candidate-set sizes."
      ],
      "limitations": [
        "Requires true structure among 256 candidates; not open-world identification.",
        "MCES uses two split variants per methods, in conflict with generic three-split caption.",
        "Separate v2 implementation and split summary from v1 measurements.",
        "No instrument-specific prospective success, calibration or clinical validity."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-msalign-v2",
          "locator": "Table 3 left panel; Section 5.1"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "baf02a4dc19b702fdad43c24f41ee8c8084ce56e4cf02bdd34e3ea0d195460fe"
    },
    {
      "id": "use-case-mapping-20260930-335-b1e55d78e4a8",
      "use_case_id": "use-case-mass-spectrum-molecule-shortlisting",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-msalign-v2-protocol-massspecgym-formula-formula-free",
      "evaluation_ids": [
        "uc20260930-msalign-v2-eval-massspecgym-formula-formula-free-deepset",
        "uc20260930-msalign-v2-eval-massspecgym-formula-formula-free-jestr",
        "uc20260930-msalign-v2-eval-massspecgym-formula-formula-free-emb-cos",
        "uc20260930-msalign-v2-eval-massspecgym-formula-formula-free-msalign",
        "uc20260930-msalign-v2-eval-massspecgym-formula-formula-free-msalign-score-fusion-overbar"
      ],
      "endpoint": "Formula-free candidate Recall@1, @5 and @20: massspecgym-formula",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "Formula-oracle comparator records excluded from this use-case mapping.",
        "Do not pool formula and MCES splits or candidate-set sizes."
      ],
      "limitations": [
        "Requires true structure among 256 candidates; not open-world identification.",
        "Three split seeds; not independently reproduced.",
        "Separate v2 implementation and split summary from v1 measurements.",
        "No instrument-specific prospective success, calibration or clinical validity."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-msalign-v2",
          "locator": "Table 3 left panel; Section 5.1"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "45da6438c305c6a8491f2cfa710afdaecbe9e1a3b56a5fb9dc330cbcea141b8f"
    },
    {
      "id": "use-case-mapping-20260930-335-c58cfe791948",
      "use_case_id": "use-case-mass-spectrum-molecule-shortlisting",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-msalign-v2-protocol-spectraverse-formula-free",
      "evaluation_ids": [
        "uc20260930-msalign-v2-eval-spectraverse-formula-free-deepset",
        "uc20260930-msalign-v2-eval-spectraverse-formula-free-jestr",
        "uc20260930-msalign-v2-eval-spectraverse-formula-free-emb-cos",
        "uc20260930-msalign-v2-eval-spectraverse-formula-free-msalign",
        "uc20260930-msalign-v2-eval-spectraverse-formula-free-msalign-score-fusion-overbar"
      ],
      "endpoint": "Formula-free candidate Recall@1, @5 and @20: spectraverse",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "Formula-oracle comparator records excluded from this use-case mapping.",
        "Do not pool formula and MCES splits or candidate-set sizes."
      ],
      "limitations": [
        "Requires true structure among 256 candidates; not open-world identification.",
        "Three split seeds; not independently reproduced.",
        "Separate v2 implementation and split summary from v1 measurements.",
        "No instrument-specific prospective success, calibration or clinical validity."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-msalign-v2",
          "locator": "Table 3 left panel; Section 5.1"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "aa1d80b28d2d2e862761c5fe867f937ccc29a92f74b8ea3299402cafb69b5b44"
    },
    {
      "id": "use-case-mapping-20260930-336-5e27f1b32dcb",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-jores-protocol-maize-protoplasts",
      "evaluation_ids": [
        "uc20260930-jores-eval-maize-protoplasts-gc-motif-linear",
        "uc20260930-jores-eval-maize-protoplasts-cnn"
      ],
      "endpoint": "Held-out promoter reporter strength, Pearson correlation squared",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "Keep maize protoplast and tobacco leaf protocols separate.",
        "No merging with AgroNT Fig3e scores."
      ],
      "limitations": [
        "Pooled species and 35S enhancer setting differ from each AgroNT species-specific comparison.",
        "No uncertainty or exact test count extracted.",
        "Reporter prediction does not establish stable-plant expression or yield."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-jores-2021",
          "locator": "Figure 8a,b; Methods: Computational modelling of promoter strength"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "d5e698a5340bc9ee1cdc240e0078c3139849de1113ee99c093139391b17706da"
    },
    {
      "id": "use-case-mapping-20260930-336-b8c3f8458c77",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-jores-protocol-tobacco-leaves",
      "evaluation_ids": [
        "uc20260930-jores-eval-tobacco-leaves-gc-motif-linear",
        "uc20260930-jores-eval-tobacco-leaves-cnn"
      ],
      "endpoint": "Held-out promoter reporter strength, Pearson correlation squared",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "Keep maize protoplast and tobacco leaf protocols separate.",
        "No merging with AgroNT Fig3e scores."
      ],
      "limitations": [
        "Pooled species and 35S enhancer setting differ from each AgroNT species-specific comparison.",
        "No uncertainty or exact test count extracted.",
        "Reporter prediction does not establish stable-plant expression or yield."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-jores-2021",
          "locator": "Figure 8a,b; Methods: Computational modelling of promoter strength"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "373c139c9ad5e5b42050548eb17765d9340b340828bd983ecc432e4e4f0a0622"
    },
    {
      "id": "use-case-mapping-20260930-337-fa51fe46268a",
      "use_case_id": "use-case-protein-stability",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-proteingym-amfr-protocol",
      "evaluation_ids": [
        "uc20260930-proteingym-amfr-eval-site-independent",
        "uc20260930-proteingym-amfr-eval-evmutation",
        "uc20260930-proteingym-amfr-eval-deepsequence-single",
        "uc20260930-proteingym-amfr-eval-deepsequence-ensemble",
        "uc20260930-proteingym-amfr-eval-eve-single",
        "uc20260930-proteingym-amfr-eval-eve-ensemble",
        "uc20260930-proteingym-amfr-eval-unirep",
        "uc20260930-proteingym-amfr-eval-unirep-evotuned",
        "uc20260930-proteingym-amfr-eval-msa-transformer-single",
        "uc20260930-proteingym-amfr-eval-msa-transformer-ensemble",
        "uc20260930-proteingym-amfr-eval-esm-1b",
        "uc20260930-proteingym-amfr-eval-esm-1v-single",
        "uc20260930-proteingym-amfr-eval-esm-1v-ensemble",
        "uc20260930-proteingym-amfr-eval-esm2-8m",
        "uc20260930-proteingym-amfr-eval-esm2-35m",
        "uc20260930-proteingym-amfr-eval-esm2-150m",
        "uc20260930-proteingym-amfr-eval-esm2-650m",
        "uc20260930-proteingym-amfr-eval-esm2-3b",
        "uc20260930-proteingym-amfr-eval-esm2-15b",
        "uc20260930-proteingym-amfr-eval-wavenet",
        "uc20260930-proteingym-amfr-eval-rita-s",
        "uc20260930-proteingym-amfr-eval-rita-m",
        "uc20260930-proteingym-amfr-eval-rita-l",
        "uc20260930-proteingym-amfr-eval-rita-xl",
        "uc20260930-proteingym-amfr-eval-progen2-s",
        "uc20260930-proteingym-amfr-eval-progen2-m",
        "uc20260930-proteingym-amfr-eval-progen2-base",
        "uc20260930-proteingym-amfr-eval-progen2-l",
        "uc20260930-proteingym-amfr-eval-progen2-xl",
        "uc20260930-proteingym-amfr-eval-gemme",
        "uc20260930-proteingym-amfr-eval-vespa",
        "uc20260930-proteingym-amfr-eval-vespal",
        "uc20260930-proteingym-amfr-eval-vespag",
        "uc20260930-proteingym-amfr-eval-protgpt2",
        "uc20260930-proteingym-amfr-eval-tranception-s-no-retrieval",
        "uc20260930-proteingym-amfr-eval-tranception-m-no-retrieval",
        "uc20260930-proteingym-amfr-eval-tranception-l-no-retrieval",
        "uc20260930-proteingym-amfr-eval-tranception-s",
        "uc20260930-proteingym-amfr-eval-tranception-m",
        "uc20260930-proteingym-amfr-eval-tranception-l",
        "uc20260930-proteingym-amfr-eval-trancepteve-s",
        "uc20260930-proteingym-amfr-eval-trancepteve-m",
        "uc20260930-proteingym-amfr-eval-trancepteve-l",
        "uc20260930-proteingym-amfr-eval-carp-38m",
        "uc20260930-proteingym-amfr-eval-carp-600k",
        "uc20260930-proteingym-amfr-eval-carp-640m",
        "uc20260930-proteingym-amfr-eval-carp-76m",
        "uc20260930-proteingym-amfr-eval-mif",
        "uc20260930-proteingym-amfr-eval-mif-st",
        "uc20260930-proteingym-amfr-eval-esm-if1",
        "uc20260930-proteingym-amfr-eval-proteinmpnn",
        "uc20260930-proteingym-amfr-eval-protssn-k-10-h-512",
        "uc20260930-proteingym-amfr-eval-protssn-k-10-h-768",
        "uc20260930-proteingym-amfr-eval-protssn-k-10-h-1280",
        "uc20260930-proteingym-amfr-eval-protssn-k-20-h-512",
        "uc20260930-proteingym-amfr-eval-protssn-k-20-h-768",
        "uc20260930-proteingym-amfr-eval-protssn-k-20-h-1280",
        "uc20260930-proteingym-amfr-eval-protssn-k-30-h-512",
        "uc20260930-proteingym-amfr-eval-protssn-k-30-h-768",
        "uc20260930-proteingym-amfr-eval-protssn-k-30-h-1280",
        "uc20260930-proteingym-amfr-eval-protssn-ensemble",
        "uc20260930-proteingym-amfr-eval-saprot-650m",
        "uc20260930-proteingym-amfr-eval-saprot-35m",
        "uc20260930-proteingym-amfr-eval-poet-200m",
        "uc20260930-proteingym-amfr-eval-mulan",
        "uc20260930-proteingym-amfr-eval-prosst-k-20",
        "uc20260930-proteingym-amfr-eval-prosst-k-128",
        "uc20260930-proteingym-amfr-eval-prosst-k-512",
        "uc20260930-proteingym-amfr-eval-prosst-k-1024",
        "uc20260930-proteingym-amfr-eval-prosst-k-2048",
        "uc20260930-proteingym-amfr-eval-prosst-k-4096",
        "uc20260930-proteingym-amfr-eval-escott",
        "uc20260930-proteingym-amfr-eval-venusrem",
        "uc20260930-proteingym-amfr-eval-rsalor",
        "uc20260930-proteingym-amfr-eval-s2f",
        "uc20260930-proteingym-amfr-eval-s2f-msa",
        "uc20260930-proteingym-amfr-eval-s3f",
        "uc20260930-proteingym-amfr-eval-s3f-msa",
        "uc20260930-proteingym-amfr-eval-siterm",
        "uc20260930-proteingym-amfr-eval-esm3-open-1-4b",
        "uc20260930-proteingym-amfr-eval-esm-c-300m",
        "uc20260930-proteingym-amfr-eval-esm-c-600m",
        "uc20260930-proteingym-amfr-eval-xtrimopglm-1b-mlm",
        "uc20260930-proteingym-amfr-eval-xtrimopglm-3b-mlm",
        "uc20260930-proteingym-amfr-eval-xtrimopglm-10b-mlm",
        "uc20260930-proteingym-amfr-eval-xtrimopglm-1b-clm",
        "uc20260930-proteingym-amfr-eval-xtrimopglm-3b-clm",
        "uc20260930-proteingym-amfr-eval-xtrimopglm-7b-clm",
        "uc20260930-proteingym-amfr-eval-xtrimopglm-100b-int4",
        "uc20260930-proteingym-amfr-eval-progen3-112m",
        "uc20260930-proteingym-amfr-eval-progen3-219m",
        "uc20260930-proteingym-amfr-eval-progen3-339m",
        "uc20260930-proteingym-amfr-eval-progen3-762m",
        "uc20260930-proteingym-amfr-eval-progen3-1b",
        "uc20260930-proteingym-amfr-eval-progen3-3b",
        "uc20260930-proteingym-amfr-eval-aido-protein-rag-16b",
        "uc20260930-proteingym-amfr-eval-protriever"
      ],
      "endpoint": "AMFR construct folding-stability rank association, official assay-level Spearman",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "Keep official source snapshot separate from local ESM2/random runs."
      ],
      "limitations": [
        "Mixed singles/doubles, not planned singles-only comparison.",
        "Assay count 2,972 does not establish complete per-model prediction coverage.",
        "No uncertainty or matched hardware measurements.",
        "Some methods require MSA or structure inputs.",
        "Construct folding stability is not clinical pathogenicity or full-protein function."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-proteingym-amfr-spearman",
          "locator": "CSV row AMFR_HUMAN_Tsuboyama_2023_4G3O; DMS metadata row"
        },
        {
          "source_id": "profile-protocol-proteingym-reference-files-dms-substitutions-csv-a8f49801",
          "locator": "CSV row AMFR_HUMAN_Tsuboyama_2023_4G3O; DMS metadata row"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "2239f0148cb569f13044f11304165339e90ca5a89a667731acd6291964041f20"
    },
    {
      "id": "use-case-mapping-20260930-338-5c7e8a1837dc",
      "use_case_id": "use-case-rhodopsin-wavelength-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-rhomax-protocol-split-3",
      "evaluation_ids": [
        "uc20260930-rhomax-eval-split-3-rhomax",
        "uc20260930-rhomax-eval-split-3-rhomax-retinal",
        "uc20260930-rhomax-eval-split-3-blasso"
      ],
      "endpoint": "Absorption-maximum prediction error in nm and eV on split 3",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "75 WT backgrounds; 884 total sequences; WT and mutants remain in one partition.",
        "BLASSO implementation adapted to eV training; not the original nm-trained benchmark.",
        "RhoMax generates AlphaFold2-derived structural binding-pocket inputs; the +retinal variant additionally includes retinal. These differ from the sequence-only FLIP2 frozen probes."
      ],
      "limitations": [
        "Different split definitions from FLIP2; scores must not be merged.",
        "WT-group holdout does not prove prospective calibration or functional utility.",
        "Printed mean row is a summary of split results, not an additional experiment.",
        "Variant denominators and split hashes remain unextracted."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-rhomax-2024",
          "locator": "Table 1; Data Set; Comparison to Previous Methods"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "16722ac202b00e28be66d43b7648909fbbbee0d04b0f95b3ce7f265568850400"
    },
    {
      "id": "use-case-mapping-20260930-338-8bb6a1bf0289",
      "use_case_id": "use-case-rhodopsin-wavelength-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-rhomax-protocol-mean",
      "evaluation_ids": [
        "uc20260930-rhomax-eval-mean-rhomax",
        "uc20260930-rhomax-eval-mean-rhomax-retinal",
        "uc20260930-rhomax-eval-mean-blasso"
      ],
      "endpoint": "Absorption-maximum prediction error in nm and eV on mean",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "75 WT backgrounds; 884 total sequences; WT and mutants remain in one partition.",
        "BLASSO implementation adapted to eV training; not the original nm-trained benchmark.",
        "RhoMax generates AlphaFold2-derived structural binding-pocket inputs; the +retinal variant additionally includes retinal. These differ from the sequence-only FLIP2 frozen probes."
      ],
      "limitations": [
        "Different split definitions from FLIP2; scores must not be merged.",
        "WT-group holdout does not prove prospective calibration or functional utility.",
        "Printed mean row is a summary of split results, not an additional experiment.",
        "Variant denominators and split hashes remain unextracted."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-rhomax-2024",
          "locator": "Table 1; Data Set; Comparison to Previous Methods"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "117a9cbf6c2385ccb88d2b7241f2c2b7d76a135f8ea2725a87b42a6f6b02a76e"
    },
    {
      "id": "use-case-mapping-20260930-338-92654511318a",
      "use_case_id": "use-case-rhodopsin-wavelength-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-rhomax-protocol-split-1",
      "evaluation_ids": [
        "uc20260930-rhomax-eval-split-1-rhomax",
        "uc20260930-rhomax-eval-split-1-rhomax-retinal",
        "uc20260930-rhomax-eval-split-1-blasso"
      ],
      "endpoint": "Absorption-maximum prediction error in nm and eV on split 1",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "75 WT backgrounds; 884 total sequences; WT and mutants remain in one partition.",
        "BLASSO implementation adapted to eV training; not the original nm-trained benchmark.",
        "RhoMax generates AlphaFold2-derived structural binding-pocket inputs; the +retinal variant additionally includes retinal. These differ from the sequence-only FLIP2 frozen probes."
      ],
      "limitations": [
        "Different split definitions from FLIP2; scores must not be merged.",
        "WT-group holdout does not prove prospective calibration or functional utility.",
        "Printed mean row is a summary of split results, not an additional experiment.",
        "Variant denominators and split hashes remain unextracted."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-rhomax-2024",
          "locator": "Table 1; Data Set; Comparison to Previous Methods"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "00408d755166f933cf3d3061ba0cd6808dd17abe43e7adb43fa4431c03fa1718"
    },
    {
      "id": "use-case-mapping-20260930-338-d2997e999587",
      "use_case_id": "use-case-rhodopsin-wavelength-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-rhomax-protocol-split-4",
      "evaluation_ids": [
        "uc20260930-rhomax-eval-split-4-rhomax",
        "uc20260930-rhomax-eval-split-4-rhomax-retinal",
        "uc20260930-rhomax-eval-split-4-blasso"
      ],
      "endpoint": "Absorption-maximum prediction error in nm and eV on split 4",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "75 WT backgrounds; 884 total sequences; WT and mutants remain in one partition.",
        "BLASSO implementation adapted to eV training; not the original nm-trained benchmark.",
        "RhoMax generates AlphaFold2-derived structural binding-pocket inputs; the +retinal variant additionally includes retinal. These differ from the sequence-only FLIP2 frozen probes."
      ],
      "limitations": [
        "Different split definitions from FLIP2; scores must not be merged.",
        "WT-group holdout does not prove prospective calibration or functional utility.",
        "Printed mean row is a summary of split results, not an additional experiment.",
        "Variant denominators and split hashes remain unextracted."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-rhomax-2024",
          "locator": "Table 1; Data Set; Comparison to Previous Methods"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "ac4f6840df24b802f0c41c9c78e176363ff4871c2fc1bf21f5d8e0b02d94af9e"
    },
    {
      "id": "use-case-mapping-20260930-338-d9d4d7039a5c",
      "use_case_id": "use-case-rhodopsin-wavelength-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-rhomax-protocol-split-2",
      "evaluation_ids": [
        "uc20260930-rhomax-eval-split-2-rhomax",
        "uc20260930-rhomax-eval-split-2-rhomax-retinal",
        "uc20260930-rhomax-eval-split-2-blasso"
      ],
      "endpoint": "Absorption-maximum prediction error in nm and eV on split 2",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "75 WT backgrounds; 884 total sequences; WT and mutants remain in one partition.",
        "BLASSO implementation adapted to eV training; not the original nm-trained benchmark.",
        "RhoMax generates AlphaFold2-derived structural binding-pocket inputs; the +retinal variant additionally includes retinal. These differ from the sequence-only FLIP2 frozen probes."
      ],
      "limitations": [
        "Different split definitions from FLIP2; scores must not be merged.",
        "WT-group holdout does not prove prospective calibration or functional utility.",
        "Printed mean row is a summary of split results, not an additional experiment.",
        "Variant denominators and split hashes remain unextracted."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-rhomax-2024",
          "locator": "Table 1; Data Set; Comparison to Previous Methods"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "ad14ab7975241d4d309e718522ac692b99dec8c00997db35c5b6a7ea1fbbaf21"
    },
    {
      "id": "use-case-mapping-20260930-339-25925e279972",
      "use_case_id": "use-case-splicing-follow-up",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "rewire-mfass-v2",
      "evaluation_ids": [
        "rewire-evaluation-baseline-kmer-position-v2",
        "rewire-local-20260921-evaluation-mfass-prior"
      ],
      "endpoint": "Historical corrected MFASS v2 ranking and top-100 follow-up precision for supervised feature and training-prior controls",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "Inspect scored and eligible counts for every condition.",
        "Consult recorded paired comparisons before interpreting differences.",
        "This mapping contains only the two baseline configurations. Historical specialist and DNABERT2 evaluations remain outside this mapping."
      ],
      "limitations": [
        "Historical v2 annotation/coverage differs from matched-annotation v1: do not combine measurements.",
        "Each method uses its own scored subset; marginal point scores do not establish a matched winner.",
        "The constant prior has tied predictions; top-100 positives depend on fixed label-independent tie order.",
        "Assay labels are not patient RNA or clinical endpoints.",
        "DNABERT2 is represented as a pipeline, which is ineligible for this exact-configuration use-case mapping. Its existing evaluation remains in the catalogue and coverage audit; no record was promoted or score duplicated."
      ],
      "citations": [
        {
          "source_id": "rewire-mfass-v2-source",
          "locator": "results/baseline-kmer-position-v2.json; matched specialist and DNABERT2 result files; training-prior.report.json"
        },
        {
          "source_id": "rewire-local-20260921-source-mfass-prior",
          "locator": "results/baseline-kmer-position-v2.json; matched specialist and DNABERT2 result files; training-prior.report.json"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "e32abf2099a69cc7d71429deec3efd1ba51921f990efcc0eab79d9e601aacb24"
    },
    {
      "id": "use-case-mapping-20260930-340-9c66a82c516b",
      "use_case_id": "use-case-utr-translation-baselines",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-framepool-protocol-random-25-100nt",
      "evaluation_ids": [
        "uc20260930-framepool-eval-random-25-100nt-optimus-50",
        "uc20260930-framepool-eval-random-25-100nt-framepool-50",
        "uc20260930-framepool-eval-random-25-100nt-optimus-100",
        "uc20260930-framepool-eval-random-25-100nt-framepool-100",
        "uc20260930-framepool-eval-random-25-100nt-framepool-combined",
        "uc20260930-framepool-eval-random-25-100nt-3mer-random-forest",
        "uc20260930-framepool-eval-random-25-100nt-4mer-random-forest",
        "uc20260930-framepool-eval-random-25-100nt-frame-forest-4mer",
        "uc20260930-framepool-eval-random-25-100nt-frame-forest-hyperopt",
        "uc20260930-framepool-eval-random-25-100nt-5mer-frame-forest"
      ],
      "endpoint": "Mean ribosome load prediction, Pearson correlation, random-25-100nt",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "Retain fixed/random, variable/random, fixed/human and variable/human panels separately."
      ],
      "limitations": [
        "Not the locally evaluated mRNABench designed-MRL split.",
        "Pearson correlation is not absolute MRL calibration or prospective design hit rate.",
        "Human 50nt/25–100nt reporter sequences are truncated; not full-length endogenous UTR validation.",
        "No per-configuration uncertainty interval is supplied in these tables."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-framepool-2021",
          "locator": "S1 Table; S4 Table for random libraries; Model training/testing"
        },
        {
          "source_id": "uc20260930-source-framepool-s1",
          "locator": "S1 Table; S4 Table for random libraries; Model training/testing"
        },
        {
          "source_id": "uc20260930-source-framepool-s4",
          "locator": "S1 Table; S4 Table for random libraries; Model training/testing"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "251cbea2f267751ecba49b69f28fdce72b07079bfa2378244d68c7738709f4d5"
    },
    {
      "id": "use-case-mapping-20260930-340-b6a5b397be2c",
      "use_case_id": "use-case-utr-translation-baselines",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-framepool-protocol-random-50nt",
      "evaluation_ids": [
        "uc20260930-framepool-eval-random-50nt-optimus-50",
        "uc20260930-framepool-eval-random-50nt-framepool-50",
        "uc20260930-framepool-eval-random-50nt-optimus-100",
        "uc20260930-framepool-eval-random-50nt-framepool-100",
        "uc20260930-framepool-eval-random-50nt-framepool-combined",
        "uc20260930-framepool-eval-random-50nt-3mer-random-forest",
        "uc20260930-framepool-eval-random-50nt-4mer-random-forest",
        "uc20260930-framepool-eval-random-50nt-frame-forest-4mer",
        "uc20260930-framepool-eval-random-50nt-frame-forest-hyperopt",
        "uc20260930-framepool-eval-random-50nt-5mer-frame-forest"
      ],
      "endpoint": "Mean ribosome load prediction, Pearson correlation, random-50nt",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "Retain fixed/random, variable/random, fixed/human and variable/human panels separately."
      ],
      "limitations": [
        "Not the locally evaluated mRNABench designed-MRL split.",
        "Pearson correlation is not absolute MRL calibration or prospective design hit rate.",
        "Human 50nt/25–100nt reporter sequences are truncated; not full-length endogenous UTR validation.",
        "No per-configuration uncertainty interval is supplied in these tables."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-framepool-2021",
          "locator": "S1 Table; S4 Table for random libraries; Model training/testing"
        },
        {
          "source_id": "uc20260930-source-framepool-s1",
          "locator": "S1 Table; S4 Table for random libraries; Model training/testing"
        },
        {
          "source_id": "uc20260930-source-framepool-s4",
          "locator": "S1 Table; S4 Table for random libraries; Model training/testing"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "35f111b81cca216d290a7e4cd30f3d3d3b85a18f7a62fd25bc48ad7b83a3fcf4"
    },
    {
      "id": "use-case-mapping-20260930-340-f41565cc8be6",
      "use_case_id": "use-case-utr-translation-baselines",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-framepool-protocol-human-50nt",
      "evaluation_ids": [
        "uc20260930-framepool-eval-human-50nt-optimus-50",
        "uc20260930-framepool-eval-human-50nt-framepool-50",
        "uc20260930-framepool-eval-human-50nt-optimus-100",
        "uc20260930-framepool-eval-human-50nt-framepool-100",
        "uc20260930-framepool-eval-human-50nt-framepool-combined"
      ],
      "endpoint": "Mean ribosome load prediction, Pearson correlation, human-50nt",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "Retain fixed/random, variable/random, fixed/human and variable/human panels separately."
      ],
      "limitations": [
        "Not the locally evaluated mRNABench designed-MRL split.",
        "Pearson correlation is not absolute MRL calibration or prospective design hit rate.",
        "Human 50nt/25–100nt reporter sequences are truncated; not full-length endogenous UTR validation.",
        "No per-configuration uncertainty interval is supplied in these tables."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-framepool-2021",
          "locator": "S1 Table; S4 Table for random libraries; Model training/testing"
        },
        {
          "source_id": "uc20260930-source-framepool-s1",
          "locator": "S1 Table; S4 Table for random libraries; Model training/testing"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "c0c42184f6e953e75c56a140c48405454050a6b673571fdd48c940ea50b80e1c"
    },
    {
      "id": "use-case-mapping-20260930-340-fe382f327cba",
      "use_case_id": "use-case-utr-translation-baselines",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc20260930-framepool-protocol-human-25-100nt",
      "evaluation_ids": [
        "uc20260930-framepool-eval-human-25-100nt-optimus-50",
        "uc20260930-framepool-eval-human-25-100nt-framepool-50",
        "uc20260930-framepool-eval-human-25-100nt-optimus-100",
        "uc20260930-framepool-eval-human-25-100nt-framepool-100",
        "uc20260930-framepool-eval-human-25-100nt-framepool-combined"
      ],
      "endpoint": "Mean ribosome load prediction, Pearson correlation, human-25-100nt",
      "relevance": "proxy",
      "rationale": "Assay-matched measured comparison informs the research decision within its original population and protocol; prospective transfer is not established.",
      "constraints": [
        "Retain fixed/random, variable/random, fixed/human and variable/human panels separately."
      ],
      "limitations": [
        "Not the locally evaluated mRNABench designed-MRL split.",
        "Pearson correlation is not absolute MRL calibration or prospective design hit rate.",
        "Human 50nt/25–100nt reporter sequences are truncated; not full-length endogenous UTR validation.",
        "No per-configuration uncertainty interval is supplied in these tables."
      ],
      "citations": [
        {
          "source_id": "uc20260930-source-framepool-2021",
          "locator": "S1 Table; S4 Table for random libraries; Model training/testing"
        },
        {
          "source_id": "uc20260930-source-framepool-s1",
          "locator": "S1 Table; S4 Table for random libraries; Model training/testing"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "850e80be452e22ffe7bb1f4dea2b4361efa6d67c29272b5e2232dde2cfdbffe9"
    },
    {
      "id": "use-case-mapping-20260930-341-146a9ab5ab18",
      "use_case_id": "use-case-rare-disease-candidate-ranking",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-talos-rgp-singleton-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-talos-rgp-singleton-default-evaluation",
        "uc-clinical-20260930-talos-rgp-singleton-strict-evaluation"
      ],
      "endpoint": "Known diagnosis recovery and number of candidates requiring review",
      "relevance": "proxy",
      "rationale": "Directly measures candidate-set recovery and review burden in a bounded clinical cohort; it does not isolate the added value of a molecular-effect model.",
      "constraints": [
        "RGP singleton; do not pool overlapping family modes.",
        "Recovery denominator is callable known diagnoses; workload denominator includes unresolved probands."
      ],
      "limitations": [
        "No prospective diagnostic accuracy or equal analyst-time comparison.",
        "Exact pipeline snapshots and independent clinical review remain outstanding."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-talos-table1",
          "locator": "Table 1, RGP, singleton"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "492fdffe33d546328398831ecbaeb15bfc4085e552a4a068324b6043eb64bb34"
    },
    {
      "id": "use-case-mapping-20260930-341-4c5f542b7511",
      "use_case_id": "use-case-rare-disease-candidate-ranking",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-exomiser-acg-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-exomiser-acg-evaluation"
      ],
      "endpoint": "Rank-budget recovery of known SNV/indel diagnoses",
      "relevance": "proxy",
      "rationale": "Measured conventional comparator for candidate ranking; separately scoped because source cohort counts conflict and effort is not matched.",
      "constraints": [
        "Results denominator 194; Extended Data Fig1 states 190.",
        "Top1/top5/top10/all rank limits are different review budgets."
      ],
      "limitations": [
        "Do not combine with Talos Table1 or infer matched equal-effort superiority.",
        "Minor software version and knowledge release not extracted."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-talos",
          "locator": "Results Comparison with other tools; Methods; Extended Data Fig1 caption"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "4ff42cdf4ab02c7e3cbe95882b79dbd9536445797deefb70b109c28fcf4662bf"
    },
    {
      "id": "use-case-mapping-20260930-341-5f740d825a8a",
      "use_case_id": "use-case-rare-disease-candidate-ranking",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-talos-rgp-full-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-talos-rgp-full-default-evaluation",
        "uc-clinical-20260930-talos-rgp-full-strict-evaluation"
      ],
      "endpoint": "Known diagnosis recovery and number of candidates requiring review",
      "relevance": "proxy",
      "rationale": "Directly measures candidate-set recovery and review burden in a bounded clinical cohort; it does not isolate the added value of a molecular-effect model.",
      "constraints": [
        "RGP full; do not pool overlapping family modes.",
        "Recovery denominator is callable known diagnoses; workload denominator includes unresolved probands."
      ],
      "limitations": [
        "No prospective diagnostic accuracy or equal analyst-time comparison.",
        "Exact pipeline snapshots and independent clinical review remain outstanding."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-talos-table1",
          "locator": "Table 1, RGP, full"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "edb1bafe60ea1e45b9648e068c9d3cad50e938a882cf87e67573f1d3c53c7d9a"
    },
    {
      "id": "use-case-mapping-20260930-341-6c73bbc682f2",
      "use_case_id": "use-case-rare-disease-candidate-ranking",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-talos-rgp-trio-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-talos-rgp-trio-default-evaluation",
        "uc-clinical-20260930-talos-rgp-trio-strict-evaluation"
      ],
      "endpoint": "Known diagnosis recovery and number of candidates requiring review",
      "relevance": "proxy",
      "rationale": "Directly measures candidate-set recovery and review burden in a bounded clinical cohort; it does not isolate the added value of a molecular-effect model.",
      "constraints": [
        "RGP trio; do not pool overlapping family modes.",
        "Recovery denominator is callable known diagnoses; workload denominator includes unresolved probands."
      ],
      "limitations": [
        "No prospective diagnostic accuracy or equal analyst-time comparison.",
        "Exact pipeline snapshots and independent clinical review remain outstanding."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-talos-table1",
          "locator": "Table 1, RGP, trio"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "95a6476b8a9b86227109e1bee8903e6707bc5fb037ec627408f01e74813a3289"
    },
    {
      "id": "use-case-mapping-20260930-341-74cc9a990fc2",
      "use_case_id": "use-case-rare-disease-candidate-ranking",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-talos-acg-singleton-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-talos-acg-singleton-default-evaluation",
        "uc-clinical-20260930-talos-acg-singleton-strict-evaluation"
      ],
      "endpoint": "Known diagnosis recovery and number of candidates requiring review",
      "relevance": "proxy",
      "rationale": "Directly measures candidate-set recovery and review burden in a bounded clinical cohort; it does not isolate the added value of a molecular-effect model.",
      "constraints": [
        "ACG singleton; do not pool overlapping family modes.",
        "Recovery denominator is callable known diagnoses; workload denominator includes unresolved probands."
      ],
      "limitations": [
        "No prospective diagnostic accuracy or equal analyst-time comparison.",
        "Exact pipeline snapshots and independent clinical review remain outstanding."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-talos-table1",
          "locator": "Table 1, ACG, singleton"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "7470a3dcdc4468ec47bbad816298725bf2bdac7221ab78556e66d404573c32e6"
    },
    {
      "id": "use-case-mapping-20260930-341-7a155a6d471d",
      "use_case_id": "use-case-rare-disease-candidate-ranking",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-talos-acg-full-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-talos-acg-full-default-evaluation",
        "uc-clinical-20260930-talos-acg-full-strict-evaluation"
      ],
      "endpoint": "Known diagnosis recovery and number of candidates requiring review",
      "relevance": "proxy",
      "rationale": "Directly measures candidate-set recovery and review burden in a bounded clinical cohort; it does not isolate the added value of a molecular-effect model.",
      "constraints": [
        "ACG full; do not pool overlapping family modes.",
        "Recovery denominator is callable known diagnoses; workload denominator includes unresolved probands."
      ],
      "limitations": [
        "No prospective diagnostic accuracy or equal analyst-time comparison.",
        "Exact pipeline snapshots and independent clinical review remain outstanding."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-talos-table1",
          "locator": "Table 1, ACG, full"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "d59140d9f8c30f8f2852c6c368e632f53b20905ad675860116f027d0d87c1764"
    },
    {
      "id": "use-case-mapping-20260930-341-882d36ea77fc",
      "use_case_id": "use-case-rare-disease-candidate-ranking",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-talos-acg-trio-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-talos-acg-trio-default-evaluation",
        "uc-clinical-20260930-talos-acg-trio-strict-evaluation"
      ],
      "endpoint": "Known diagnosis recovery and number of candidates requiring review",
      "relevance": "proxy",
      "rationale": "Directly measures candidate-set recovery and review burden in a bounded clinical cohort; it does not isolate the added value of a molecular-effect model.",
      "constraints": [
        "ACG trio; do not pool overlapping family modes.",
        "Recovery denominator is callable known diagnoses; workload denominator includes unresolved probands."
      ],
      "limitations": [
        "No prospective diagnostic accuracy or equal analyst-time comparison.",
        "Exact pipeline snapshots and independent clinical review remain outstanding."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-talos-table1",
          "locator": "Table 1, ACG, trio"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "723379867421472cc78c45e6c184d220b5d72a73bbed537bccb847fa33d66f0c"
    },
    {
      "id": "use-case-mapping-20260930-342-3c7f6c5e1873",
      "use_case_id": "use-case-unresolved-rare-disease-reanalysis",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-reanalysis-programme-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-reanalysis-programme-evaluation"
      ],
      "endpoint": "New confirmed diagnoses and review workload during monthly reanalysis",
      "relevance": "direct",
      "rationale": "Direct programme-level reanalysis outcomes in a previously unresolved cohort; does not isolate method benefit.",
      "constraints": [
        "4735 individuals; diagnoses and individuals distinct; staggered repeated cycles.",
        "Original analysis 2017–2022; monthly programme 2023–2025."
      ],
      "limitations": [
        "No equal-input refreshed conventional control.",
        "No randomization/blinding; updated calls and knowledge contribute to yield."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-talos",
          "locator": "Results, Iterative reanalysis; Methods"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "d887ee28c6c1f3c874a767d7bb2ec1231494ac199987506d9b9f426a494d4dbb"
    },
    {
      "id": "use-case-mapping-20260930-343-2b9e45a46d4b",
      "use_case_id": "use-case-brca1-brca2-germline-interpretation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-enigma-brca1-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-enigma-brca1-evaluation"
      ],
      "endpoint": "Functional-reference likelihood ratios for computational evidence weights",
      "relevance": "proxy",
      "rationale": "A gene-specific evidence input calibration relevant to BRCA review; it does not measure full clinical classification or review effort.",
      "constraints": [
        "Domain-restricted missense variants.",
        "Functionally defined labels and calibration-selected cutpoints."
      ],
      "limitations": [
        "Independent pathogenicity labels, training overlap audit and serious-error adjudication remain missing.",
        "No method-level clinical superiority claim."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-enigma",
          "locator": "Table 2 and footnotes a–c"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "c3fd4bc1f567201eacfaa792450dec9142a84b3e2e7727b66ea693aabfa50dfa"
    },
    {
      "id": "use-case-mapping-20260930-343-64fb29b5bff8",
      "use_case_id": "use-case-brca1-brca2-germline-interpretation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-enigma-brca2-generic-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-enigma-brca2-generic-evaluation"
      ],
      "endpoint": "Evidence-code assignment under generic thresholds",
      "relevance": "proxy",
      "rationale": "Published conventional threshold comparison on the functional benign reference set.",
      "constraints": [
        "Primary narrative 29% BRCA1 / 36% BRCA2 false PP3; BP4 coverage<10%is a bound, not a point estimate."
      ],
      "limitations": [
        "No final classification error or independent clinical labels.",
        "Full Supplementary Table S4 unavailable."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-enigma",
          "locator": "Results, BayesDel calibration paragraph comparing Pejaver thresholds"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "ee252deea607df5ce7d4f609c7eecdd1a7f7360e86603a45f2b10aa0fac2bf40"
    },
    {
      "id": "use-case-mapping-20260930-343-77a677a9f119",
      "use_case_id": "use-case-brca1-brca2-germline-interpretation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-enigma-brca1-generic-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-enigma-brca1-generic-evaluation"
      ],
      "endpoint": "Evidence-code assignment under generic thresholds",
      "relevance": "proxy",
      "rationale": "Published conventional threshold comparison on the functional benign reference set.",
      "constraints": [
        "Primary narrative 29% BRCA1 / 36% BRCA2 false PP3; BP4 coverage<10%is a bound, not a point estimate."
      ],
      "limitations": [
        "No final classification error or independent clinical labels.",
        "Full Supplementary Table S4 unavailable."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-enigma",
          "locator": "Results, BayesDel calibration paragraph comparing Pejaver thresholds"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "10af50994eb7dc1629ea5928300ae2c5afafede588f1a4ff87dc8fa530fd8757"
    },
    {
      "id": "use-case-mapping-20260930-343-cf70c367d48e",
      "use_case_id": "use-case-brca1-brca2-germline-interpretation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-enigma-brca2-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-enigma-brca2-evaluation"
      ],
      "endpoint": "Functional-reference likelihood ratios for computational evidence weights",
      "relevance": "proxy",
      "rationale": "A gene-specific evidence input calibration relevant to BRCA review; it does not measure full clinical classification or review effort.",
      "constraints": [
        "Domain-restricted missense variants.",
        "Functionally defined labels and calibration-selected cutpoints."
      ],
      "limitations": [
        "Independent pathogenicity labels, training overlap audit and serious-error adjudication remain missing.",
        "No method-level clinical superiority claim."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-enigma",
          "locator": "Table 2 and footnotes a–c"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "642adb9c1bc9d3ec6ea06911f920eb4ea874af3b3a719285f1895c431c35e20a"
    },
    {
      "id": "use-case-mapping-20260930-343-dd4044ac3f2f",
      "use_case_id": "use-case-brca1-brca2-germline-interpretation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-enigma-pilot-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-enigma-pilot-evaluation"
      ],
      "endpoint": "Selected-variant curation agreement and uncertainty resolution",
      "relevance": "proxy",
      "rationale": "Measures part of the actual BRCA review workflow, without independent clinical accuracy or effort validation.",
      "constraints": [
        "40 selected pilot variants; two phases; v1.0 specifications."
      ],
      "limitations": [
        "Same pilot supports specification revision.",
        "No blinded independent error adjudication or review-time comparator."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-enigma",
          "locator": "Results Pilot curation; Figure 3"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "7befed0b2dbb3b4996da72cba55c5bb55eaa94fc343a3b3dd75ae585aceff920"
    },
    {
      "id": "use-case-mapping-20260930-344-401bec55db2f",
      "use_case_id": "use-case-somatic-small-variant-oncogenicity",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-oncovi-expert-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-oncovi-expert-evaluation",
        "uc-clinical-20260930-oncovi-prior-mtb-expert-evaluation"
      ],
      "endpoint": "Oncogenicity category agreement, sensitivity and reported subgroup performance",
      "relevance": "proxy",
      "rationale": "Assesses variant classification within the named reference; reference independence and transfer are bounded.",
      "constraints": [
        "A posteriori disagreement-enriched sample; same three experts and same reference resources; blinded to implementation, not independent resource adjudication.",
        "Historical MTB 34% agreement with expert review compares functional-impact and oncogenicity labels; OncoVI 22% is against the old MTB labels, while 80% is against expert reassessment. Distinct metrics and references must not be pooled."
      ],
      "limitations": [
        "Not a treatment-selection benchmark.",
        "No human domain review of this intake; reference overlap and prospective workload remain unresolved.",
        "Abstract/body accuracy discrepancies retained."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-oncovi",
          "locator": "Results; Figures 2–4; ClinVar assessment; subgroup values referenced to Supplemental Table S6"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "5d61748874a6f679048f652c596002cdd63e8b5c32c8dd12a2c2804a4389fb5b"
    },
    {
      "id": "use-case-mapping-20260930-344-8840aa223ede",
      "use_case_id": "use-case-somatic-small-variant-oncogenicity",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-oncovi-mtb-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-oncovi-mtb-evaluation"
      ],
      "endpoint": "Oncogenicity category agreement, sensitivity and reported subgroup performance",
      "relevance": "proxy",
      "rationale": "Assesses variant classification within the named reference; reference independence and transfer are bounded.",
      "constraints": [
        "Prior MTB effect-on-protein-function classes are a proxy reference, not oncogenicity ground truth."
      ],
      "limitations": [
        "Not a treatment-selection benchmark.",
        "No human domain review of this intake; reference overlap and prospective workload remain unresolved.",
        "Abstract/body accuracy discrepancies retained."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-oncovi",
          "locator": "Results; Figures 2–4; ClinVar assessment; subgroup values referenced to Supplemental Table S6"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "b1e4888eb2945bc0acb969a826916d7400b02343fbf8a45918a868418d732c35"
    },
    {
      "id": "use-case-mapping-20260930-344-93b84bc9cf35",
      "use_case_id": "use-case-somatic-small-variant-oncogenicity",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-oncovi-clinvar-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-oncovi-clinvar-evaluation"
      ],
      "endpoint": "Oncogenicity category agreement, sensitivity and reported subgroup performance",
      "relevance": "direct",
      "rationale": "Assesses variant classification within the named reference; reference independence and transfer are bounded.",
      "constraints": [
        "691 classified variants in 227 genes, downloaded 12 April 2025; predictor/curation overlap needs audit."
      ],
      "limitations": [
        "Not a treatment-selection benchmark.",
        "No human domain review of this intake; reference overlap and prospective workload remain unresolved.",
        "Abstract/body accuracy discrepancies retained."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-oncovi",
          "locator": "Results; Figures 2–4; ClinVar assessment; subgroup values referenced to Supplemental Table S6"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "f614d64f1f4d0651e3da8e14e0a6c88948e14b98bf4efcebeadbef9347f3d283"
    },
    {
      "id": "use-case-mapping-20260930-344-a19f78d8cc9b",
      "use_case_id": "use-case-somatic-small-variant-oncogenicity",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-oncovi-sop-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-oncovi-sop-evaluation"
      ],
      "endpoint": "Oncogenicity category agreement, sensitivity and reported subgroup performance",
      "relevance": "direct",
      "rationale": "Assesses variant classification within the named reference; reference independence and transfer are bounded.",
      "constraints": [
        "Three-class O/LO, VUS, B/LB reference; public guideline examples, not unseen-gene validation."
      ],
      "limitations": [
        "Not a treatment-selection benchmark.",
        "No human domain review of this intake; reference overlap and prospective workload remain unresolved.",
        "Abstract/body accuracy discrepancies retained."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-oncovi",
          "locator": "Results; Figures 2–4; ClinVar assessment; subgroup values referenced to Supplemental Table S6"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "3d0eb0760cc7a4313cd1cc46ead5f252f9c67a85036c3e22a3061b1d5465a425"
    },
    {
      "id": "use-case-mapping-20260930-345-ba1ec69965eb",
      "use_case_id": "use-case-egfr-nsclc-actionability-resistance-evidence",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "uc-clinical-20260930-civic-retrieval-protocol",
      "evaluation_ids": [
        "uc-clinical-20260930-civic-mcp-evaluation",
        "uc-clinical-20260930-civic-alone-evaluation",
        "uc-clinical-20260930-civic-agent-evaluation"
      ],
      "endpoint": "CIViC evidence-direction retrieval precision/recall/F1 and latency",
      "relevance": "proxy",
      "rationale": "A directly relevant retrieval endpoint across cancer types, not a validated EGFR/NSCLC evidence-review system.",
      "constraints": [
        "100 triplets; March2026 live CIViC; fixed label prompts.",
        "Exact API model and separate unpinnable UI Agent Mode.",
        "No Evidence excluded from reported performance."
      ],
      "limitations": [
        "No EGFR-specific subgroup, equal-effort manual baseline, treatment-line history or jurisdiction adjudication.",
        "Figure1c and supplement disagree for MCP oncogenic F1.",
        "No individual benefit or treatment-selection claim."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20260930-source-civic-supp",
          "locator": "Supplementary Tables 1–3, implementation and Agent Mode methods; main Fig1c"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "87704e39bcce294a0e4e2aa7f06674b10169e976a057f55b805d09bd08423041"
    },
    {
      "id": "use-case-mapping-20260930-346-15b6eed3fb55",
      "use_case_id": "use-case-therapeutic-target-validation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "ucc-research-protocol-cppc-challenge2-original",
      "evaluation_ids": [
        "ucc-research-eval-cppc-table-s2-d-row-2",
        "ucc-research-eval-cppc-table-s2-d-row-3",
        "ucc-research-eval-cppc-table-s2-d-row-4",
        "ucc-research-eval-cppc-table-s2-d-row-5",
        "ucc-research-eval-cppc-table-s2-d-row-6",
        "ucc-research-eval-cppc-table-s2-d-row-7",
        "ucc-research-eval-cppc-table-s2-d-row-11",
        "ucc-research-eval-cppc-table-s2-d-row-12",
        "ucc-research-eval-cppc-table-s2-d-row-13",
        "ucc-research-eval-cppc-table-s2-d-row-14",
        "ucc-research-eval-cppc-table-s2-d-row-15",
        "ucc-research-eval-cppc-table-s2-d-row-16",
        "ucc-research-eval-cppc-table-s2-d-row-17",
        "ucc-research-eval-cppc-table-s2-d-row-18",
        "ucc-research-eval-cppc-table-s2-d-row-19",
        "ucc-research-eval-cppc-table-s2-d-row-20",
        "ucc-research-eval-cppc-table-s2-d-row-21",
        "ucc-research-eval-cppc-table-s2-d-row-22",
        "ucc-research-eval-cppc-table-s2-d-row-23",
        "ucc-research-eval-cppc-table-s2-d-row-24"
      ],
      "endpoint": "Prospective target ranking: custom top-k overlap AUC under source-defined objective and filtering aggregation.",
      "relevance": "proxy",
      "rationale": "Evaluates original submissions against measured state objectives in the nominated Screen2 target universe.",
      "constraints": [
        "Test universe selected by top Challenge1 nominations, not a representative genome-wide holdout.",
        "Different from original held-out response prediction and from post-hoc best reimplementations."
      ],
      "limitations": [
        "Custom ranking AUC is not ROC AUC, prospective hit rate or efficacy.",
        "Original model implementations/checkpoints are not pinned in the intake; no uncertainty printed.",
        "Automated source review only; independent human scientific review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "ucc-research-source-challenge-paper",
          "locator": "Table S2 sheet Challenge 1 & 2 column D; Challenge setup and results"
        },
        {
          "source_id": "ucc-research-source-challenge-supp-s2",
          "locator": "Table S2 sheet Challenge 1 & 2 column D; Challenge setup and results"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "8ef3a3a3cb5440ad3de91bfc38479b8a81ee20340244f85d7a4fee93f849f7ea"
    },
    {
      "id": "use-case-mapping-20260930-346-910458dfd409",
      "use_case_id": "use-case-therapeutic-target-validation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "ucc-research-protocol-cppc-prospective-selection",
      "evaluation_ids": [
        "ucc-research-eval-cppc-nominated-target-outcome"
      ],
      "endpoint": "Prospective nomination campaign: count of named targets improving the defined T-cell state objective.",
      "relevance": "proxy",
      "rationale": "Desired T-cell state changes provide experimentally measured target nomination evidence, but target efficacy/selectivity/rescue remain outside the measured endpoint.",
      "constraints": [
        "Pooled nominations; not individual-method scores.",
        "Mouse B16-OVA melanoma OT1 CD8 T-cell context."
      ],
      "limitations": [
        "No matched-budget conventional/random prospective comparison; no antitumor efficacy, rescue or selectivity result in this endpoint.",
        "61 nominations / Results 57 selected targets / README 59 targets / 50 post-QC perturbations need reconciliation; no success fraction calculated.",
        "Preprint v1; original and reimplemented ranking scores kept separate. Screen2 ranking population is enriched by original nomination methods.",
        "Automated source review only; independent human scientific review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "ucc-research-source-challenge-paper",
          "locator": "Abstract; Challenge 2; Figure 5; README"
        },
        {
          "source_id": "ucc-research-source-challenge-code",
          "locator": "Abstract; Challenge 2; Figure 5; README"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "67aa800c66ccf8b230c977e9998254c9fa0706a1a40a6929b05b31bfc635e58d"
    },
    {
      "id": "use-case-mapping-20260930-347-81e84acc2274",
      "use_case_id": "use-case-regulatory-variant-gene-follow-up",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "ucc-research-protocol-mprabc-k562",
      "evaluation_ids": [
        "ucc-research-eval-mprabc-full",
        "ucc-research-eval-mprabc-sei",
        "ucc-research-eval-mprabc-mpralegnet",
        "ucc-research-eval-mprabc-re2g",
        "ucc-research-eval-mprabc-abc",
        "ucc-research-eval-mprabc-megamap"
      ],
      "endpoint": "K562 CRISPRi enhancer–gene linking AUPRC and precision at 70% recall.",
      "relevance": "proxy",
      "rationale": "Endogenous whole-element perturbation links support one component of locus follow-up; no allele-editing or disease causality claim.",
      "constraints": [
        "10,356 pairs, 471 positive, 9,885 negative; bootstrap intervals use 10,000 pair resamples."
      ],
      "limitations": [
        "Training/test independence unresolved: paper explicitly says benchmarking pipeline does not perform cross-validation.",
        "No allele-specific endogenous-edit benchmark or equal-budget prospective shortlist benefit established.",
        "Automated source review only; independent human scientific review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "ucc-research-source-mprabc",
          "locator": "Table 3; Benchmarking; Figure 2 Results"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "ba830f4f42b704d185e6a54460c08de0e4fc11b7bfeb9006404dab1b0d1af669"
    },
    {
      "id": "use-case-mapping-20260930-348-15b6eed3fb55",
      "use_case_id": "use-case-phenotype-perturbation-selection",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "ucc-research-protocol-cppc-challenge2-original",
      "evaluation_ids": [
        "ucc-research-eval-cppc-table-s2-d-row-2",
        "ucc-research-eval-cppc-table-s2-d-row-3",
        "ucc-research-eval-cppc-table-s2-d-row-4",
        "ucc-research-eval-cppc-table-s2-d-row-5",
        "ucc-research-eval-cppc-table-s2-d-row-6",
        "ucc-research-eval-cppc-table-s2-d-row-7",
        "ucc-research-eval-cppc-table-s2-d-row-11",
        "ucc-research-eval-cppc-table-s2-d-row-12",
        "ucc-research-eval-cppc-table-s2-d-row-13",
        "ucc-research-eval-cppc-table-s2-d-row-14",
        "ucc-research-eval-cppc-table-s2-d-row-15",
        "ucc-research-eval-cppc-table-s2-d-row-16",
        "ucc-research-eval-cppc-table-s2-d-row-17",
        "ucc-research-eval-cppc-table-s2-d-row-18",
        "ucc-research-eval-cppc-table-s2-d-row-19",
        "ucc-research-eval-cppc-table-s2-d-row-20",
        "ucc-research-eval-cppc-table-s2-d-row-21",
        "ucc-research-eval-cppc-table-s2-d-row-22",
        "ucc-research-eval-cppc-table-s2-d-row-23",
        "ucc-research-eval-cppc-table-s2-d-row-24"
      ],
      "endpoint": "Prospective target ranking: custom top-k overlap AUC under source-defined objective and filtering aggregation.",
      "relevance": "direct",
      "rationale": "Evaluates original submissions against measured state objectives in the nominated Screen2 target universe.",
      "constraints": [
        "Test universe selected by top Challenge1 nominations, not a representative genome-wide holdout.",
        "Different from original held-out response prediction and from post-hoc best reimplementations."
      ],
      "limitations": [
        "Custom ranking AUC is not ROC AUC, prospective hit rate or efficacy.",
        "Original model implementations/checkpoints are not pinned in the intake; no uncertainty printed.",
        "Automated source review only; independent human scientific review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "ucc-research-source-challenge-paper",
          "locator": "Table S2 sheet Challenge 1 & 2 column D; Challenge setup and results"
        },
        {
          "source_id": "ucc-research-source-challenge-supp-s2",
          "locator": "Table S2 sheet Challenge 1 & 2 column D; Challenge setup and results"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "8ef3a3a3cb5440ad3de91bfc38479b8a81ee20340244f85d7a4fee93f849f7ea"
    },
    {
      "id": "use-case-mapping-20260930-348-910458dfd409",
      "use_case_id": "use-case-phenotype-perturbation-selection",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "ucc-research-protocol-cppc-prospective-selection",
      "evaluation_ids": [
        "ucc-research-eval-cppc-nominated-target-outcome"
      ],
      "endpoint": "Prospective nomination campaign: count of named targets improving the defined T-cell state objective.",
      "relevance": "direct",
      "rationale": "Direct experimental follow-up of predicted interventions for a specified state objective; narrow mouse T-cell setting.",
      "constraints": [
        "Pooled nominations; not individual-method scores.",
        "Mouse B16-OVA melanoma OT1 CD8 T-cell context."
      ],
      "limitations": [
        "No matched-budget conventional/random prospective comparison; no antitumor efficacy, rescue or selectivity result in this endpoint.",
        "61 nominations / Results 57 selected targets / README 59 targets / 50 post-QC perturbations need reconciliation; no success fraction calculated.",
        "Preprint v1; original and reimplemented ranking scores kept separate. Screen2 ranking population is enriched by original nomination methods.",
        "Automated source review only; independent human scientific review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "ucc-research-source-challenge-paper",
          "locator": "Abstract; Challenge 2; Figure 5; README"
        },
        {
          "source_id": "ucc-research-source-challenge-code",
          "locator": "Abstract; Challenge 2; Figure 5; README"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "67aa800c66ccf8b230c977e9998254c9fa0706a1a40a6929b05b31bfc635e58d"
    },
    {
      "id": "use-case-mapping-20260930-349-1af09c2e0e0f",
      "use_case_id": "use-case-cell-type-annotation-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "ucc-research-protocol-sctab-seed",
      "evaluation_ids": [
        "ucc-research-eval-sctab-seed-sctab",
        "ucc-research-eval-sctab-seed-xgboost",
        "ucc-research-eval-sctab-seed-mlp",
        "ucc-research-eval-sctab-seed-linear",
        "ucc-research-eval-sctab-seed-celltypist",
        "ucc-research-eval-sctab-seed-scgpt-zero",
        "ucc-research-eval-sctab-seed-scgpt-ft",
        "ucc-research-eval-sctab-seed-ciform",
        "ucc-research-eval-sctab-seed-uce"
      ],
      "endpoint": "Donor-held-out, ontology-adjusted macro F1 for known cell types.",
      "relevance": "proxy",
      "rationale": "Direct annotation endpoint, but only partial evidence for transfer to a new study/platform and unsupported populations.",
      "constraints": [
        "10x-related assays; rare cell types filtered; donor holdouts, not whole-study holdouts.",
        "CellTypist/scGPT/CIForm training subsampling differs from full-data models."
      ],
      "limitations": [
        "Unknown-type rejection ROC appears in Supplementary Figure 4 but numerical curve labels are not extracted here.",
        "Reference labels are author annotations, not independent ground truth.",
        "Checkpoint and split hashes unextracted; no deployment calibration claim.",
        "Automated source review only; independent human scientific review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "ucc-research-source-sctab-paper",
          "locator": "Supplementary Table 1a; Results and Methods"
        },
        {
          "source_id": "ucc-research-source-sctab-supp",
          "locator": "Supplementary Table 1a; Results and Methods"
        },
        {
          "source_id": "ucc-research-source-sctab-code",
          "locator": "Supplementary Table 1a; Results and Methods"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "4872ccaea598db25c6371bbcfe66915403991bf2f182b54c0a3e8ed2b337cfbc"
    },
    {
      "id": "use-case-mapping-20260930-350-54e8c47f1c29",
      "use_case_id": "use-case-structural-hypotheses-experiments",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "ucc-docking-cluspro-bm5-2020-protocol-others-top10",
      "evaluation_ids": [
        "ucc-docking-cluspro-bm5-2020-eval-others-top10"
      ],
      "endpoint": "At least one DockQ≥0.23 docking pose in the top10 cluster centers for others BM5 complexes",
      "relevance": "proxy",
      "rationale": "A measured conventional structural-input docking baseline for selecting interface hypotheses; separate from sequence-based cofolding and experimental binding validation.",
      "constraints": [
        "One of102 original others targets could not be evaluated by DockQ.",
        "Input component 3D structures, not just sequence; default category-specific mode.",
        "Top10 models ranked by cluster population.",
        "Enzyme and others categories remain separate; antibody and aggregate rows excluded from #350."
      ],
      "limitations": [
        "Cannot compare numerical differences directly to FoldBench because datasets, structures and protocols differ.",
        "A successful docking pose does not establish binding, affinity, mechanism or cellular interaction.",
        "One of102 others targets was unevaluable; scores use101.",
        "No model execution or human domain review; exact software revision unextracted."
      ],
      "citations": [
        {
          "source_id": "ucc-docking-cluspro-bm5-2020-source",
          "locator": "Table1 Others first row; Results; STAR Methods"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "c3e3836140d1b003a4a2cfa35fe104c24b99977065ad025ff21b63e76256f143"
    },
    {
      "id": "use-case-mapping-20260930-350-746ab300cdfe",
      "use_case_id": "use-case-structural-hypotheses-experiments",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "ucc-docking-cluspro-bm5-2020-protocol-enzyme-top30",
      "evaluation_ids": [
        "ucc-docking-cluspro-bm5-2020-eval-enzyme-top30"
      ],
      "endpoint": "At least one DockQ≥0.23 docking pose in the top30 cluster centers for enzyme BM5 complexes",
      "relevance": "proxy",
      "rationale": "A measured conventional structural-input docking baseline for selecting interface hypotheses; separate from sequence-based cofolding and experimental binding validation.",
      "constraints": [
        "Component structures downloaded from BM5; docking uses balanced coefficients.",
        "Input component 3D structures, not just sequence; default category-specific mode.",
        "Top30 models ranked by cluster population.",
        "Enzyme and others categories remain separate; antibody and aggregate rows excluded from #350."
      ],
      "limitations": [
        "Cannot compare numerical differences directly to FoldBench because datasets, structures and protocols differ.",
        "A successful docking pose does not establish binding, affinity, mechanism or cellular interaction.",
        "All88 enzyme targets evaluated.",
        "No model execution or human domain review; exact software revision unextracted."
      ],
      "citations": [
        {
          "source_id": "ucc-docking-cluspro-bm5-2020-source",
          "locator": "Table1 Enzyme second row; Results; STAR Methods"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "a3a301cf0076ee932dcf378bda828dc1b185d161611f8c23d5f33eaf613a8b9d"
    },
    {
      "id": "use-case-mapping-20260930-350-85c0ffafe7cd",
      "use_case_id": "use-case-structural-hypotheses-experiments",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "ucc-docking-cluspro-bm5-2020-protocol-enzyme-top10",
      "evaluation_ids": [
        "ucc-docking-cluspro-bm5-2020-eval-enzyme-top10"
      ],
      "endpoint": "At least one DockQ≥0.23 docking pose in the top10 cluster centers for enzyme BM5 complexes",
      "relevance": "proxy",
      "rationale": "A measured conventional structural-input docking baseline for selecting interface hypotheses; separate from sequence-based cofolding and experimental binding validation.",
      "constraints": [
        "Component structures downloaded from BM5; docking uses balanced coefficients.",
        "Input component 3D structures, not just sequence; default category-specific mode.",
        "Top10 models ranked by cluster population.",
        "Enzyme and others categories remain separate; antibody and aggregate rows excluded from #350."
      ],
      "limitations": [
        "Cannot compare numerical differences directly to FoldBench because datasets, structures and protocols differ.",
        "A successful docking pose does not establish binding, affinity, mechanism or cellular interaction.",
        "All88 enzyme targets evaluated.",
        "No model execution or human domain review; exact software revision unextracted."
      ],
      "citations": [
        {
          "source_id": "ucc-docking-cluspro-bm5-2020-source",
          "locator": "Table1 Enzyme first row; Results; STAR Methods"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "f8a47cbbffb9b66a7667c211d3a21ac10666d1c6970672b363f728aaba7eac22"
    },
    {
      "id": "use-case-mapping-20260930-350-ae963dc25ccb",
      "use_case_id": "use-case-structural-hypotheses-experiments",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "ucc-docking-cluspro-bm5-2020-protocol-others-top30",
      "evaluation_ids": [
        "ucc-docking-cluspro-bm5-2020-eval-others-top30"
      ],
      "endpoint": "At least one DockQ≥0.23 docking pose in the top30 cluster centers for others BM5 complexes",
      "relevance": "proxy",
      "rationale": "A measured conventional structural-input docking baseline for selecting interface hypotheses; separate from sequence-based cofolding and experimental binding validation.",
      "constraints": [
        "One of102 original others targets could not be evaluated by DockQ.",
        "Input component 3D structures, not just sequence; default category-specific mode.",
        "Top30 models ranked by cluster population.",
        "Enzyme and others categories remain separate; antibody and aggregate rows excluded from #350."
      ],
      "limitations": [
        "Cannot compare numerical differences directly to FoldBench because datasets, structures and protocols differ.",
        "A successful docking pose does not establish binding, affinity, mechanism or cellular interaction.",
        "One of102 others targets was unevaluable; scores use101.",
        "No model execution or human domain review; exact software revision unextracted."
      ],
      "citations": [
        {
          "source_id": "ucc-docking-cluspro-bm5-2020-source",
          "locator": "Table1 Others second row; Results; STAR Methods"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "48c2b0dfa68d8cddc7d714e510f921d1becd5a86f8a914405b4e27fdde694252"
    },
    {
      "id": "use-case-mapping-20260930-350-d7a6088b3a59",
      "use_case_id": "use-case-structural-hypotheses-experiments",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the 17-use-case coverage audit; retain narrower endpoint and transfer limitations.",
      "protocol_id": "ucc-research-protocol-foldbench-protein-protein",
      "evaluation_ids": [
        "ucc-research-eval-foldbench-protein-protein-af3",
        "ucc-research-eval-foldbench-protein-protein-boltz1",
        "ucc-research-eval-foldbench-protein-protein-chai1",
        "ucc-research-eval-foldbench-protein-protein-helixfold3",
        "ucc-research-eval-foldbench-protein-protein-protenix"
      ],
      "endpoint": "Task-specific top-ranked structure accuracy and success metrics.",
      "relevance": "proxy",
      "rationale": "Structure/pose accuracy can inform experimental hypotheses; measured utility of mutation or construct choices is absent.",
      "constraints": [
        "protein-protein only; total 279; assessable per model [279, 264, 264, 265, 266].",
        "5 seeds × 5 samples, 10 recycles, top model-ranked prediction; pinned inference commits."
      ],
      "limitations": [
        "Scored target populations differ; Table 1 assessable counts are retained as coverage context, not verified metric denominators. Some Table 3 rates do not reconcile with integer counts after rounding. No complete-cohort estimate is inferred.",
        "No intervals in Table 3; no prospective experimental utility or matched conventional mutation/construct baseline.",
        "Automated source review only; independent human scientific review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "ucc-research-source-foldbench-paper",
          "locator": "Supplementary Tables 1–3; Model Inference; Evaluation and metrics"
        },
        {
          "source_id": "ucc-research-source-foldbench-supp",
          "locator": "Supplementary Tables 1–3; Model Inference; Evaluation and metrics"
        },
        {
          "source_id": "ucc-research-source-foldbench-code",
          "locator": "Supplementary Tables 1–3; Model Inference; Evaluation and metrics"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex source curation, independent worker cross-review and root integration review",
        "reviewed_at": "2026-09-30T21:54:30.217Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "721c5b13cb48a83b53c6b366a3b8b3422b4613ccba7a5f957fd08a23f4b4eb41"
    },
    {
      "id": "use-case-mapping-20261005-343-1d76c37513fa",
      "use_case_id": "use-case-brca1-brca2-germline-interpretation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the additive 2026-10-05 BRCA1/BRCA2 evidence intake (issue #343); scope the endpoint to tool-vs-tool agreement on conflicting BRCA1 missense variants, not accuracy.",
      "protocol_id": "uc-clinical-20261005-so-2024-protocol",
      "evaluation_ids": [
        "uc-clinical-20261005-so-2024-evaluation"
      ],
      "endpoint": "Tool-vs-tool concordance between VarSome and the CanVIG-UK BRCA1/BRCA2 gene-specific guidance, each consolidated to a three-category scheme, on 450 BRCA1 missense variants with conflicting ClinVar interpretations",
      "relevance": "proxy",
      "rationale": "Measures agreement between two interpretation tools on a conflicting-classification subset; this is tool-vs-tool agreement on disputed BRCA1 missense variants, not accuracy against independent ground truth.",
      "constraints": [
        "450 BRCA1 missense variants with conflicting ClinVar interpretations as of 20 December 2022."
      ],
      "limitations": [
        "Tool-vs-tool agreement on conflicting-classification variants, not accuracy against independent ground truth.",
        "The source's ground-truth-anchored comparator covers only 17/450 (3.8%) of variants and is not used in this mapping.",
        "No reviewer time or serious-error adjudication reported."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20261005-source-so-2024",
          "locator": "Results, Section 3.3, Comparison of Classification Results Between Varsome and CanVIG-UK"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet clinical coverage worker (research-evidence intake), root integration review",
        "reviewed_at": "2026-10-05T17:39:47.679Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks against the additive 2026-10-05 BRCA1/BRCA2 evidence intake (issue #343). No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "48f65d049d907742ed728f2473bdabcabf49b7bcc26bc8be979614ee88e306f9"
    },
    {
      "id": "use-case-mapping-20261005-343-4d6887d8d2e1",
      "use_case_id": "use-case-brca1-brca2-germline-interpretation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the additive 2026-10-05 BRCA1/BRCA2 evidence intake (issue #343); scope the endpoint to functional-data classification yield in BRCA2 exons 15-26, not full professional review.",
      "protocol_id": "uc-clinical-20261005-hu-2026-protocol",
      "evaluation_ids": [
        "uc-clinical-20261005-hu-2026-evaluation"
      ],
      "endpoint": "Fraction of 6,383 BRCA2 exon 15-26 single-nucleotide variants reaching a final pathogenic/likely pathogenic or benign/likely benign classification using combined saturation-genome-editing functional-data evidence (Integrated VarCall model) under the ENIGMA BRCA1/BRCA2 VCEP specification",
      "relevance": "proxy",
      "rationale": "A functional-data classification yield/coverage figure for a combined SGE assay model applied to a single BRCA2 domain; it measures how many variants reach a final call under integrated functional evidence, not independent clinical accuracy and not a complete professional classification review.",
      "constraints": [
        "6,383 BRCA2 single-nucleotide variants, exons 15-26 (C-terminal DNA-binding domain) only; not representative of the full gene or of BRCA1."
      ],
      "limitations": [
        "Classification yield/coverage using an integrated functional-data proxy, not independent clinical accuracy and not a complete professional classification review.",
        "BRCA2 only, limited to exons 15-26; matched six-model comparator panels and all per-model sensitivity/specificity figures are not used in this mapping.",
        "No reviewer time or independent clinical adjudication reported."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20261005-source-hu-2026",
          "locator": "Results, Incorporation of the Integrated VarCall model functional data into the BRCA1/2 ClinGen variant classification specifications"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet clinical coverage worker (research-evidence intake), root integration review",
        "reviewed_at": "2026-10-05T17:39:47.679Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks against the additive 2026-10-05 BRCA1/BRCA2 evidence intake (issue #343). No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "16b62ca6230dd2a624f367524a71ae47084254b3cd85b2c960950ce4a55db9d4"
    },
    {
      "id": "use-case-mapping-20261005-343-6620c2b7cd1a",
      "use_case_id": "use-case-brca1-brca2-germline-interpretation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the additive 2026-10-05 BRCA1/BRCA2 evidence intake (issue #343); scope the endpoint to paired criteria-version reclassification, not accuracy or review time.",
      "protocol_id": "uc-clinical-20261005-benet-pages-t2-protocol",
      "evaluation_ids": [
        "uc-clinical-20261005-benet-pages-t2-evaluation"
      ],
      "endpoint": "Fraction of a fixed 121-variant BRCA1/BRCA2 VUS cohort reclassified to likely benign under the ACMG/AMP classification system with current SVI recommendations and new annotation data (t2)",
      "relevance": "proxy",
      "rationale": "A paired criteria-version reclassification step on a fixed VUS cohort; it measures how many variants move off VUS under a later criteria set, not classification accuracy against independent ground truth or reviewer time.",
      "constraints": [
        "Same 121-variant VUS cohort reclassified across successive criteria-set versions (t1-t2-t3); this mapping covers only the ACMG/AMP + SVI step."
      ],
      "limitations": [
        "Paired criteria-version reclassification rate, not accuracy against independent ground truth or a reviewer-time comparison.",
        "No independent clinical adjudication of individual calls.",
        "The source's unresolved BRCA1/BRCA2 gene-split sentence (85% n=40 / 83% n=67) is not used in this mapping."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20261005-source-benet-pages",
          "locator": "Results, Reclassification of variants using the ACMG/AMP classification system with SVI recommendations and new data (ACMG/AMP + SVI)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet clinical coverage worker (research-evidence intake), root integration review",
        "reviewed_at": "2026-10-05T17:39:47.679Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks against the additive 2026-10-05 BRCA1/BRCA2 evidence intake (issue #343). No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "d90ea87913d66bf0b1188b3fc78a1d452e47b77baef4901c1e6339f111f52a9f"
    },
    {
      "id": "use-case-mapping-20261005-343-9a16ffea921e",
      "use_case_id": "use-case-brca1-brca2-germline-interpretation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the additive 2026-10-05 BRCA1/BRCA2 evidence intake (issue #343); scope the endpoint to agreement with eRepo submissions from an unreviewed preprint, not independent accuracy. Corrected 2026-10-06: the comparator was re-verified live against the preprint and relabelled from \"ClinVar expert-panel\" to the ENIGMA VCEP's expert-curated classifications and evidence-code assignments in the ClinGen Evidence Repository (eRepo); the 108/143 and 326/413 figures are unchanged.",
      "protocol_id": "uc-clinical-20261005-hector-erepo-protocol",
      "evaluation_ids": [
        "uc-clinical-20261005-hector-erepo-evaluation"
      ],
      "endpoint": "Classification-level and evidence-code-level agreement between the HECTOR automated classifier and the ENIGMA VCEP's expert-curated calls on the 143-variant ClinGen Evidence Repository (eRepo) reference set",
      "relevance": "proxy",
      "rationale": "An unreviewed preprint's measured agreement between an automated classifier and ClinGen eRepo expert-panel submissions; this is agreement with eRepo submissions, not independent accuracy, and the result is not peer reviewed.",
      "constraints": [
        "medRxiv preprint v1, posted 2026-07-06; not peer reviewed.",
        "143-variant ENIGMA Evidence Repository reference set; 413 evidence codes."
      ],
      "limitations": [
        "Unreviewed preprint: agreement with the ENIGMA VCEP's expert-curated classifications and evidence-code assignments in the ClinGen Evidence Repository (eRepo), not independent accuracy against separately adjudicated ground truth.",
        "Not peer reviewed; no independent adjudication of either the automated classifier's or the expert panel's calls.",
        "No reviewer time measured."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20261005-source-hector-preprint",
          "locator": "Results, Evidence Repository"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet clinical coverage worker (research-evidence intake), root integration review",
        "reviewed_at": "2026-10-06T09:38:06.000Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks against the additive 2026-10-05 BRCA1/BRCA2 evidence intake (issue #343). No new model execution, independent experimental replication, qualified human scientific review or clinical validation. Corrected 2026-10-06: re-verified live against the preprint full text and relabelled the eRepo comparator; evidence_sha256 refreshed to match the corrected protocol and result records."
      },
      "evidence_sha256": "fe8cb49219f859f42dbedc89cd877a9ed7e120be6b8b9c48c07ab251e594357d"
    },
    {
      "id": "use-case-mapping-20261005-343-b3489257a74c",
      "use_case_id": "use-case-brca1-brca2-germline-interpretation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add independently checked primary-source protocol evidence from the additive 2026-10-05 BRCA1/BRCA2 evidence intake (issue #343); scope the endpoint to paired criteria-version reclassification, not accuracy or review time.",
      "protocol_id": "uc-clinical-20261005-benet-pages-t3-protocol",
      "evaluation_ids": [
        "uc-clinical-20261005-benet-pages-t3-evaluation"
      ],
      "endpoint": "Fraction of the same 121-variant BRCA1/BRCA2 VUS cohort reclassified to benign/likely benign under the ENIGMA BRCA1/BRCA2 VCEP specification v1.1.0 (t3), applied after the ACMG/AMP + SVI step",
      "relevance": "proxy",
      "rationale": "A paired criteria-version reclassification step on the same fixed VUS cohort, applied after t2; it measures how many variants move off VUS under the ENIGMA VCEP specification, not classification accuracy against independent ground truth or reviewer time.",
      "constraints": [
        "Same 121-variant VUS cohort reclassified across successive criteria-set versions (t1-t2-t3); this mapping covers only the ENIGMA VCEP step, applied after t2 on the same cohort."
      ],
      "limitations": [
        "Paired criteria-version reclassification rate, not accuracy against independent ground truth or a reviewer-time comparison.",
        "No independent clinical adjudication of individual calls.",
        "The source's unresolved BRCA1/BRCA2 gene-split sentence (85% n=40 / 83% n=67) is not used in this mapping."
      ],
      "citations": [
        {
          "source_id": "uc-clinical-20261005-source-benet-pages",
          "locator": "Results, Reclassification of variants using the ENIGMA specifications (ENIGMA VCEP)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet clinical coverage worker (research-evidence intake), root integration review",
        "reviewed_at": "2026-10-05T17:39:47.679Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks against the additive 2026-10-05 BRCA1/BRCA2 evidence intake (issue #343). No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "33b1fe66003215802f6cc3f25de57a51d59312c12960386c4aefbca20a0bc3cf"
    },
    {
      "id": "use-case-mapping-20261006-349-2cfbe5c7674a",
      "use_case_id": "use-case-cell-type-annotation-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add scTab's own unknown/absent-type detection ROC-AUC (deep-ensemble uncertainty score, correct known-type vs. absent-type), independently source-checked in the additive 2026-10-06 intake (issue #349). Kept as a separate mapping from the existing Table 1a known-type macro-F1 mapping and from scTab's own known-type error-detection ROC-AUC, which must not be conflated with unknown-type detection.",
      "protocol_id": "ucc-research-protocol-sctab-uncertainty-absent",
      "evaluation_ids": [
        "ucc-research-eval-sctab-uncertainty-absent"
      ],
      "endpoint": "Deep-ensemble (5-model) uncertainty-score ROC-AUC distinguishing correctly predicted known cell types from cell types entirely absent from training, within one donor-held-out split of scTab's pooled 249-dataset CELLxGENE corpus.",
      "relevance": "proxy",
      "rationale": "Directly on the use case's unknown-type-recognition endpoint, but single-corpus and donor-level only; no cross-study or cross-platform transfer of this capability is established by this source.",
      "constraints": [
        "Donor-level holdout within one pooled 249-dataset corpus (10x-related assays only; rare types excluded); not a whole-study or whole-platform holdout. Dataset/study overlap between training and test splits is not directly confirmed in the retrieved text.",
        "Single ROC-AUC value; no rejection-at-coverage curve or demonstrated calibration."
      ],
      "limitations": [
        "This is unknown/absent-type detection only; it must not be combined with or treated as evidence for scTab's separate known-type error-detection ROC-AUC (0.891, see use-case-mapping-20261006-349-6a9d63555f43).",
        "Exact number of cell types and cells comprising the absent-type comparison group is not stated in the retrieved main text; held here as an explicit missing denominator, not invented.",
        "Reference/training labels are author annotations, not independent ground truth.",
        "Automated source review only; independent human scientific review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "ucc-research-source-sctab-paper",
          "locator": "Methods, \"Uncertainty quantification for scTab model\""
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet cell-type-annotation-transfer evidence-research worker",
        "reviewed_at": "2026-10-06T22:39:40Z",
        "note": "Source-backed primary-text transcription of the 2026-10-06 cell-type-annotation-transfer research dossier (docs/omics/evidence-research/cell-type-annotation-transfer-2026-10-06.md), corrected after an independent source check; bound by data/omics/use-case-coverage-20261006/review.json. Revised same day after a second independent review corrected the uncertainty-averaging order, removed an unsupported 'temperature-scaled' claim, replaced n_runs with ensemble_size, and removed an unsupported 'data augmentation enabled' assertion; evidence_sha256 refreshed accordingly. No ROC-AUC value or endpoint scoping changed. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "a1690c50b7f8265bfb8f0cac670bc262b7174377da5556a1edfae508abb68189"
    },
    {
      "id": "use-case-mapping-20261006-349-6a9d63555f43",
      "use_case_id": "use-case-cell-type-annotation-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add scTab's own known-type error/confidence-detection ROC-AUC (deep-ensemble uncertainty score, correct vs. incorrect known-type), independently source-checked in the additive 2026-10-06 intake (issue #349). Kept as a separate mapping from the existing Table 1a known-type macro-F1 mapping and from scTab's own unknown/absent-type detection ROC-AUC, which must not be conflated with this known-type endpoint.",
      "protocol_id": "ucc-research-protocol-sctab-uncertainty-error",
      "evaluation_ids": [
        "ucc-research-eval-sctab-uncertainty-error"
      ],
      "endpoint": "Deep-ensemble (5-model) uncertainty-score ROC-AUC distinguishing correctly from incorrectly predicted known cell types, within one donor-held-out split of scTab's pooled 249-dataset CELLxGENE corpus.",
      "relevance": "proxy",
      "rationale": "Relevant to the use case's \"identify cells that require expert review\" endpoint for known-type calls (flagging the model's own low-confidence/incorrect predictions among trained types), but it is NOT an unknown-type or absent-reference detection result, and no cross-study or cross-platform transfer of this capability is established by this source.",
      "constraints": [
        "Donor-level holdout within one pooled 249-dataset corpus (10x-related assays only; rare types excluded); not a whole-study or whole-platform holdout.",
        "Single ROC-AUC value; no rejection-at-coverage curve or demonstrated calibration."
      ],
      "limitations": [
        "This is known-type error/confidence detection, not unknown-type or absent-reference detection; it must not be combined with or treated as evidence for scTab's separate absent-type ROC-AUC (0.782, see use-case-mapping-20261006-349-2cfbe5c7674a).",
        "Exact number of cells comprising the incorrect-known-type comparison group is not stated in the retrieved main text; held here as an explicit missing denominator, not invented.",
        "Reference/training labels are author annotations, not independent ground truth.",
        "Automated source review only; independent human scientific review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "ucc-research-source-sctab-paper",
          "locator": "Methods, \"Uncertainty quantification for scTab model\""
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet cell-type-annotation-transfer evidence-research worker",
        "reviewed_at": "2026-10-06T22:39:40Z",
        "note": "Source-backed primary-text transcription of the 2026-10-06 cell-type-annotation-transfer research dossier (docs/omics/evidence-research/cell-type-annotation-transfer-2026-10-06.md), corrected after an independent source check; bound by data/omics/use-case-coverage-20261006/review.json. Revised same day after a second independent review corrected the uncertainty-averaging order, removed an unsupported 'temperature-scaled' claim, replaced n_runs with ensemble_size, and removed an unsupported 'data augmentation enabled' assertion; evidence_sha256 refreshed accordingly. No ROC-AUC value or endpoint scoping changed. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "e58579cd7ad115270a2313f48faa46a431b3598e4e85e07d1b8fe13756627255"
    },
    {
      "id": "use-case-mapping-20261006-349-967caca02e7d",
      "use_case_id": "use-case-cell-type-annotation-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Abdelaal et al. 2019 Baron Human INTRA-dataset rejection-option baseline (same-day continuation of the additive 2026-10-06 intake, issue #349): four median F1 values and three cells-left-unlabeled percentages for conventional classifiers, re-verified directly against the cached primary source. This is explicitly NOT a cross-study or cross-platform result; the source's own genuine inter-dataset experiments (Fig. 4, Fig. 5) are not ingested here.",
      "protocol_id": "ucc-research-protocol-abdelaal-baron-human-intra",
      "evaluation_ids": [
        "ucc-research-eval-abdelaal-baron-human-svm",
        "ucc-research-eval-abdelaal-baron-human-svm-rejection",
        "ucc-research-eval-abdelaal-baron-human-scmapcell",
        "ucc-research-eval-abdelaal-baron-human-scpred"
      ],
      "endpoint": "Intra-dataset (same single Baron Human dataset, stratified 5-fold cross-validation, identical folds across classifiers) median F1-score and, for rejection-capable classifiers, percentage of cells left unlabeled.",
      "relevance": "proxy",
      "rationale": "A conventional-classifier rejection-option baseline directly relevant to the use case's 'identify cells that require expert review' endpoint, but intra-dataset only; it does not establish cross-study or cross-platform transfer, and the compared classifiers (SVM, SVMrejection, scmapcell, scPred) predate scTab, CellTypist, Azimuth, scANVI and scParadise.",
      "constraints": [
        "INTRA-dataset (same single dataset, stratified 5-fold CV, identical folds across classifiers); NOT cross-study or cross-platform. Confirmed from the source's own Methods, which explicitly distinguishes this design from a separate inter-dataset design.",
        "Baron Human dataset only: 8,569 cells, 17,499 genes, 14 cell populations (13 after <10-cell filtering), inDrop protocol (Table 2)."
      ],
      "limitations": [
        "Does not establish cross-study or cross-platform transfer; the source's own inter-dataset (cross-platform/cross-study) experiments (Fig. 4 brain, Fig. 5 pancreatic) are not ingested in this mapping.",
        "Median F1-score is a per-dataset summary statistic across cell populations, not a per-population breakdown.",
        "SVM (no rejection) is quoted as classifying \"100% of the cells\" (0% unlabeled by implication) but is not given its own rejection-percentage result record in this intake.",
        "Reference labels are the original Baron et al. study's author-assigned cell-population annotations, not independently adjudicated ground truth.",
        "Exact per-fold cell counts and the fold manifest are not stated in the retrieved main text.",
        "8,569 cells is the Table 2 Baron Human dataset-size figure; it is not independently confirmed as the exact number of cells scored for any individual classifier's median F1 or unlabeled-percentage result.",
        "Percentage unlabeled is a coverage/rejection-rate tradeoff against the same classifier's own median F1-score, not a standalone performance figure with a universal better/worse direction; it must be read jointly with that classifier's F1, not in isolation. The corresponding result records carry metric_direction: \"unknown\" for this reason, per the schema's higher/lower/unknown contract.",
        "Automated source review only; independent human scientific review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "ucc-research-source-abdelaal-2019-paper",
          "locator": "Results, \"All classifiers perform well in intra-dataset experiments\"; Methods, \"Intra-dataset classification\"; Fig. 1a,b; Table 1; Table 2"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet cell-type-annotation-transfer evidence-research worker",
        "reviewed_at": "2026-10-06T23:32:17Z",
        "note": "Source-backed primary-text transcription of Abdelaal et al. 2019, continuing the bounded 2026-10-06 cell-type-annotation-transfer research dossier and same-day additive intake; bound by data/omics/use-case-coverage-20261006/review.json. Revised after an independent review: the source's retrieved_at is now the real re-fetch timestamp 2026-10-06T23:28:54Z UTC (byte-identical; a bounded-window estimate was used previously), its bytes are archived as a committed artifact (not only the gitignored workbench copy), the three unlabeled-percentage results' metric_direction is now \"unknown\" (was \"lower\", an unsupported universal claim) with an explicit F1/rejection-tradeoff note, and the 8,569-cell figure is now explicitly scoped as the Table 2 dataset size, not a confirmed scored denominator. No printed value changed; evidence_sha256 refreshed accordingly. scArches's ~84% accuracy figure was re-checked in the same pass and found unresolved (accuracy-formula denominator not stated for the reporting figure); no record for it exists in this release. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "e0f5ec5a720c706fbe9c2cee13e1079f53b36c874c3ba392c7baee4165690e55"
    },
    {
      "id": "use-case-mapping-20261007-345-7c4af3e091bd",
      "use_case_id": "use-case-egfr-nsclc-actionability-resistance-evidence",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add a second pan-cancer proxy: CIViC-Fact v3's post-cutoff within-linked-publication passage-retrieval evaluation. Independently reviewed by Codex across four review cycles before ingestion; retains the existing CIViC MCP proxy mapping unchanged.",
      "protocol_id": "ucc-clinical-egfr-protocol-civicfact-v3-retrieval",
      "evaluation_ids": [
        "ucc-clinical-egfr-eval-medcpt-ft-postcutoff",
        "ucc-clinical-egfr-eval-qwen3-8b-postcutoff"
      ],
      "endpoint": "CIViC-Fact v3 post-cutoff within-linked-publication passage-retrieval appropriate-content rate (manual review), two retriever configurations on the identical 40-entry cohort",
      "relevance": "proxy",
      "rationale": "A pan-cancer, within-known-publication passage-retrieval evaluation; not a validated EGFR/NSCLC evidence-retrieval system, and not a source-search or citation-context benchmark across an open literature corpus.",
      "constraints": [
        "150 candidate CIViC entries submitted/revised 2026-03-03 to 2026-06-09; single temporal cohort, not independent replication.",
        "40/150 scored after sequential exclusions (68 full-text inaccessible, 14 already present in train/dev/test data, 2 data errors, 24 NEI requiring supplementary/image evidence, 2 NEI substantially revised claims); partial, enriched-cohort coverage.",
        "Manual review judgment of 'appropriate' retrieved content, not an automated or recomputed metric; reviewer count, blinding and replicate/seed structure for this judgment are unreported, and uncertainty is unreported."
      ],
      "limitations": [
        "No EGFR-specific subgroup or score anywhere in this cohort.",
        "Enriched cohort per the source's own Discussion; not representative of CIViC overall.",
        "Retrieval ranks passages within each entry's already-linked source publication, not open-corpus literature search.",
        "Does not close the benchmark-execution gap tracked separately at rewire-benchmarks #28.",
        "Not the static v2 SUPPORTS/REFUTES/NEI stance-classification accuracy (89%/85%) or the v2-era BGE retriever manual-review figure (70/75); those are distinct, version-specific measurements and are not represented by this mapping."
      ],
      "citations": [
        {
          "source_id": "ucc-clinical-egfr-source-civicfact-v3",
          "locator": "Results, 'Passage Retrieval Models Perform Well without Fine-Tuning' subsection (printed p.14 / PDF p.15); selection chain in Methods, 'Temporal evaluation' subsection and Supplementary Figure 2 (printed p.33 / PDF p.34)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet EGFR-evidence-research worker, independently reviewed by Codex across four review cycles",
        "reviewed_at": "2026-10-07T10:26:56.000Z",
        "note": "Source-backed literature curation with independent automated transcription and scope checks. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "73429814be7f5d20503b6cce41f30f875d995c497f97cb4d22987c30f1a14d90"
    },
    {
      "id": "use-case-mapping-20261007-349-b12e5dc4f5b4",
      "use_case_id": "use-case-cell-type-annotation-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Bounded additive intake (issue #349), continuing the 2026-10-06/2026-10-07 dossier series, independently reviewed by Codex before ingestion: Abdelaal et al. 2019 Figure S10 Panel B (34-population inter-dataset, cross-dataset/cross-species-where-applicable brain comparison), SVMrejection row only (9 values). VISp and ALM share GSE115746. MTG is a separate human single-nucleus dataset; combinations involving MTG include cross-species data. Some two-dataset training combinations still share the mouse study with the test dataset, so independent-study holdout must not be inferred for all nine combinations. A conventional rejection-option baseline, not a known-type accuracy or unknown-type-detection accuracy claim, and not a superiority claim across methods. Kept as a separate mapping from the existing Baron Human INTRA-dataset mapping (a different, non-transfer design) and from the two scTab uncertainty mappings.",
      "protocol_id": "ucc-research-protocol-abdelaal-brain-inter-34pop",
      "evaluation_ids": [
        "ucc-research-eval-abdelaal-brain-alm-from-visp",
        "ucc-research-eval-abdelaal-brain-alm-from-mtg",
        "ucc-research-eval-abdelaal-brain-alm-from-visp-mtg",
        "ucc-research-eval-abdelaal-brain-mtg-from-visp",
        "ucc-research-eval-abdelaal-brain-mtg-from-alm",
        "ucc-research-eval-abdelaal-brain-mtg-from-visp-alm",
        "ucc-research-eval-abdelaal-brain-visp-from-alm",
        "ucc-research-eval-abdelaal-brain-visp-from-mtg",
        "ucc-research-eval-abdelaal-brain-visp-from-alm-mtg"
      ],
      "endpoint": "Inter-dataset (cross-dataset, cross-species where applicable) SVMrejection conventional rejection-option baseline: percentage of test-set cells left unlabeled across all 9 VISp/ALM/MTG train-test combinations, 34-cell-population annotation level.",
      "relevance": "proxy",
      "rationale": "A conventional-classifier rejection-option baseline on a genuine cross-dataset brain annotation-transfer design. VISp and ALM share GSE115746. MTG is a separate human single-nucleus dataset; combinations involving MTG include cross-species data. Some two-dataset training combinations still share the mouse study with the test dataset, so independent-study holdout must not be inferred for all nine combinations. This is directly relevant to the use case's 'identify cells that require expert review' endpoint. It is NOT a known-type accuracy result, NOT an unknown/absent-type-detection accuracy result, and NOT a comparison against any other classifier (only SVMrejection is ingested); it predates and does not evaluate scTab, CellTypist, Azimuth, scANVI or scParadise.",
      "constraints": [
        "Inter-dataset (whole-dataset train/test combinations across VISp, ALM, MTG), not a within-dataset cross-validation split.",
        "34-cell-population (deeper) annotation level only (Figure S10 Panel B); the 3-population major-lineage level (Figure S10 Panel A) is not included.",
        "SVMrejection only; the other 17 classifiers shown in the same source figure are not included in this mapping."
      ],
      "limitations": [
        "Scored denominator is UNREPORTED in the retrieved main text and supplement. Table 2's raw per-dataset cell counts (VISp 12,832; ALM 8,758; MTG 14,636) are recorded on the linked dataset records as context only, not a confirmed denominator, and must not be used to back-calculate an implied rejected-cell count.",
        "metric_direction is \"unknown\" for every result under this mapping's protocol: percentage unlabeled is a coverage/rejection-rate figure, not a standalone performance metric with one universal better/worse direction.",
        "Not a known-type accuracy claim, not an unknown/absent-type-detection accuracy claim, and not a superiority claim over any other classifier; no comparison across methods is asserted by this mapping.",
        "Figure S9 Panel A (PBMC) carries no printed digit labels in the source PDF at any resolution checked, and separately shows an unresolved conflict between its own caption (\"median F1-score\"), the page's only legend (\"Unlabeled (%)\"), and the figure's overall title (\"Percentage of unlabeled cells\"); nothing from Figure S9 is part of this mapping.",
        "Reference labels are the original brain-atlas studies' author-assigned cell-population annotations, not independently adjudicated ground truth.",
        "VISp and ALM share GSE115746. MTG is a separate human single-nucleus dataset; combinations involving MTG include cross-species data. Some two-dataset training combinations still share the mouse study with the test dataset, so independent-study holdout must not be inferred for all nine combinations.",
        "Automated source review only; independent human scientific review remains outstanding.",
        "Predates scTab, CellTypist, Azimuth, scANVI and scParadise; does not evaluate them. Does not measure or substitute for the common-protocol cross-study benchmark gap tracked separately at rewire-benchmarks #27."
      ],
      "citations": [
        {
          "source_id": "ucc-research-source-abdelaal-2019-supplement",
          "locator": "Figure S10, Panel B (page 15 of 18), row \"SVMrejection\""
        },
        {
          "source_id": "ucc-research-source-abdelaal-2019-paper",
          "locator": "Methods, \"Brain\" (train-test design); Table 2 (VISp/ALM/MTG raw dataset sizes, context only)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet cell-type-annotation-transfer evidence-research worker",
        "reviewed_at": "2026-10-07T07:44:08Z",
        "note": "Source-backed primary-text transcription of Abdelaal et al. 2019 Figure S10 Panel B, bound by data/omics/use-case-coverage-20261007/research/review.json. Independently cross-checked against the original supplement PDF by a separate reviewer (Codex) before this mapping was added, including the nine printed values, column order, and the Figure S9 Panel A caption/legend/title conflict and Table 2 denominator-scope correction carried over from the preceding 2026-10-07 completion-assessment review. No new model execution, independent experimental replication, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "2c244a1d6cdfa0b4059e49cb8b72580fefa55effe869edbbb507aa8dc33a68cc"
    },
    {
      "id": "use-case-mapping-amp-20261007-feng-pathogenic-common-variant",
      "use_case_id": "use-case-rare-disease-candidate-ranking",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-feng-20261007-protocol-pathogenic-common-variant-classification",
      "evaluation_ids": [
        "amp-feng-20261007-eval-variant-sei-hidden-states",
        "amp-feng-20261007-eval-variant-sei-output-tracks",
        "amp-feng-20261007-eval-variant-enformer-hidden-states",
        "amp-feng-20261007-eval-variant-enformer-output-tracks",
        "amp-feng-20261007-eval-variant-dnabert-2",
        "amp-feng-20261007-eval-variant-nt-v2",
        "amp-feng-20261007-eval-variant-hyenadna",
        "amp-feng-20261007-eval-variant-hyenadna-450k-long-sequence",
        "amp-feng-20261007-eval-variant-caduceus-ph",
        "amp-feng-20261007-eval-variant-caduceus-ph-long-sequence",
        "amp-feng-20261007-eval-variant-grover"
      ],
      "endpoint": "Pathogenic-versus-common SNP discrimination (AUC, Cohen's d) for eleven DNA foundation-model/genomic-comparator configurations, across separate short- and long-window scored populations",
      "relevance": "proxy",
      "rationale": "Genome-wide pathogenic-versus-common SNP discrimination is a proxy for rare-disease candidate variant ranking; it is not disease-specific (no BRCA subgroup claim) and is not a complete clinical classification.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "No disease-specific (e.g. BRCA) subgroup claim is supported by this table.",
        "Short-window and long-window populations are separate and must not be pooled; each configuration/dataset records its exact assignment.",
        "Sei, Enformer (hidden/output tracks) are non-DNA-foundation specialised genomic comparators per the source's own asterisk notation, not DNA foundation models.",
        "No independent reproduction or qualified human scientific review."
      ],
      "citations": [
        {
          "source_id": "evidence-expansion-dna-foundation-models-2025-5d8ca9bc",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "2d0b549dcee81e4cfed5e2a0aafe4eb1db610a1ee23da5e18dcd2bc38e3913ce"
    },
    {
      "id": "use-case-mapping-amp-20261007-feng-promoter-nontata",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-feng-20261007-protocol-promoter-nontata",
      "evaluation_ids": [
        "amp-feng-20261007-eval-promoter-nontata-dnabert-2",
        "amp-feng-20261007-eval-promoter-nontata-nt-v2",
        "amp-feng-20261007-eval-promoter-nontata-hyenadna",
        "amp-feng-20261007-eval-promoter-nontata-caduceus-ph",
        "amp-feng-20261007-eval-promoter-nontata-grover"
      ],
      "endpoint": "Promoter Arabidopsis NonTATA sequence-classification AUC for five DNA foundation-model embeddings",
      "relevance": "proxy",
      "rationale": "Sequence-label NonTATA promoter classification AUC is a proxy for reporter-measured promoter strength; it is a separate task from TATA classification and does not itself measure reporter expression.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Sequence classification accuracy is not a reporter expression measurement.",
        "Exact per-task split sizes/counts are unextracted from the inspected main text (Supplementary Data 6 is identified but not inspected).",
        "No independent reproduction or qualified human scientific review."
      ],
      "citations": [
        {
          "source_id": "evidence-expansion-dna-foundation-models-2025-5d8ca9bc",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "484019f2b00fc14a76245b942ee2284d896ade9ac7ae1881ffb57f2d48c74646"
    },
    {
      "id": "use-case-mapping-amp-20261007-feng-promoter-tata",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-feng-20261007-protocol-promoter-tata",
      "evaluation_ids": [
        "amp-feng-20261007-eval-promoter-tata-dnabert-2",
        "amp-feng-20261007-eval-promoter-tata-nt-v2",
        "amp-feng-20261007-eval-promoter-tata-hyenadna",
        "amp-feng-20261007-eval-promoter-tata-caduceus-ph",
        "amp-feng-20261007-eval-promoter-tata-grover"
      ],
      "endpoint": "Promoter Arabidopsis TATA sequence-classification AUC for five DNA foundation-model embeddings",
      "relevance": "proxy",
      "rationale": "Sequence-label TATA promoter classification AUC is a proxy for reporter-measured promoter strength; it is a separate task from NonTATA classification and does not itself measure reporter expression.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Sequence classification accuracy is not a reporter expression measurement.",
        "Exact per-task split sizes/counts are unextracted from the inspected main text (Supplementary Data 6 is identified but not inspected).",
        "No independent reproduction or qualified human scientific review."
      ],
      "citations": [
        {
          "source_id": "evidence-expansion-dna-foundation-models-2025-5d8ca9bc",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "82cdfa5a836cb8dbca37e6b7358371b345cff6715624503b26b80c14ffd99e6b"
    },
    {
      "id": "use-case-mapping-amp-20261007-feng-qtl-eqtl",
      "use_case_id": "use-case-regulatory-variant-gene-follow-up",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-feng-20261007-protocol-qtl-eqtl",
      "evaluation_ids": [
        "amp-feng-20261007-eval-qtl-eqtl-sei-hidden-states",
        "amp-feng-20261007-eval-qtl-eqtl-sei-output-tracks",
        "amp-feng-20261007-eval-qtl-eqtl-enformer-hidden-states",
        "amp-feng-20261007-eval-qtl-eqtl-enformer-output-tracks",
        "amp-feng-20261007-eval-qtl-eqtl-dnabert-2",
        "amp-feng-20261007-eval-qtl-eqtl-nt-v2",
        "amp-feng-20261007-eval-qtl-eqtl-hyenadna",
        "amp-feng-20261007-eval-qtl-eqtl-hyenadna-450k-long-sequence",
        "amp-feng-20261007-eval-qtl-eqtl-caduceus-ph",
        "amp-feng-20261007-eval-qtl-eqtl-caduceus-ph-long-sequence",
        "amp-feng-20261007-eval-qtl-eqtl-grover",
        "amp-feng-20261007-eval-qtl-eqtl-alphagenome-output-tracks"
      ],
      "endpoint": "eQTL causal-versus-noncausal variant discrimination (AUC, Cohen's d) for twelve DNA foundation-model/genomic-comparator configurations",
      "relevance": "proxy",
      "rationale": "eQTL causal-variant discrimination is a proxy for regulatory-variant gene follow-up prioritisation; it is not an experimentally validated follow-up choice and is a separate task from the other three QTL types.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Post-window-filter per-model scored denominators are unreported for eQTL; the pre-filter positive count (1896) is not a confirmed denominator.",
        "AlphaGenome appears only in this QTL comparison, not the pathogenic/common SNP comparison. It uses the long-window QTL dataset, consuming central 131,072 bp and averaging output tracks over central 2,048 bp.",
        "Sei, Enformer (hidden/output tracks) and AlphaGenome (output tracks) are non-DNA-foundation specialised genomic comparators per the source's own asterisk notation, not DNA foundation models.",
        "No independent reproduction or qualified human scientific review."
      ],
      "citations": [
        {
          "source_id": "evidence-expansion-dna-foundation-models-2025-5d8ca9bc",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "d0b9b6fefcc9ddecfd669a085afc1a750f8e9abea41282aa5fb29905c825e713"
    },
    {
      "id": "use-case-mapping-amp-20261007-feng-qtl-ipaqtl",
      "use_case_id": "use-case-regulatory-variant-gene-follow-up",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-feng-20261007-protocol-qtl-ipaqtl",
      "evaluation_ids": [
        "amp-feng-20261007-eval-qtl-ipaqtl-sei-hidden-states",
        "amp-feng-20261007-eval-qtl-ipaqtl-sei-output-tracks",
        "amp-feng-20261007-eval-qtl-ipaqtl-enformer-hidden-states",
        "amp-feng-20261007-eval-qtl-ipaqtl-enformer-output-tracks",
        "amp-feng-20261007-eval-qtl-ipaqtl-dnabert-2",
        "amp-feng-20261007-eval-qtl-ipaqtl-nt-v2",
        "amp-feng-20261007-eval-qtl-ipaqtl-hyenadna",
        "amp-feng-20261007-eval-qtl-ipaqtl-hyenadna-450k-long-sequence",
        "amp-feng-20261007-eval-qtl-ipaqtl-caduceus-ph",
        "amp-feng-20261007-eval-qtl-ipaqtl-caduceus-ph-long-sequence",
        "amp-feng-20261007-eval-qtl-ipaqtl-grover",
        "amp-feng-20261007-eval-qtl-ipaqtl-alphagenome-output-tracks"
      ],
      "endpoint": "ipaQTL causal-versus-noncausal variant discrimination (AUC, Cohen's d) for twelve DNA foundation-model/genomic-comparator configurations",
      "relevance": "proxy",
      "rationale": "ipaQTL causal-variant discrimination is a proxy for regulatory-variant gene follow-up prioritisation; it is not an experimentally validated follow-up choice and is a separate task from the other three QTL types.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Post-window-filter per-model scored denominators are unreported for ipaQTL; the pre-filter positive count (116) is not a confirmed denominator.",
        "AlphaGenome appears only in this QTL comparison, not the pathogenic/common SNP comparison. It uses the long-window QTL dataset, consuming central 131,072 bp and averaging output tracks over central 2,048 bp.",
        "Sei, Enformer (hidden/output tracks) and AlphaGenome (output tracks) are non-DNA-foundation specialised genomic comparators per the source's own asterisk notation, not DNA foundation models.",
        "No independent reproduction or qualified human scientific review."
      ],
      "citations": [
        {
          "source_id": "evidence-expansion-dna-foundation-models-2025-5d8ca9bc",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "7da1452b791f1146c42d650a409cbc8728a649b4cecc04322d25d420c2ebdaef"
    },
    {
      "id": "use-case-mapping-amp-20261007-feng-qtl-paqtl",
      "use_case_id": "use-case-regulatory-variant-gene-follow-up",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-feng-20261007-protocol-qtl-paqtl",
      "evaluation_ids": [
        "amp-feng-20261007-eval-qtl-paqtl-sei-hidden-states",
        "amp-feng-20261007-eval-qtl-paqtl-sei-output-tracks",
        "amp-feng-20261007-eval-qtl-paqtl-enformer-hidden-states",
        "amp-feng-20261007-eval-qtl-paqtl-enformer-output-tracks",
        "amp-feng-20261007-eval-qtl-paqtl-dnabert-2",
        "amp-feng-20261007-eval-qtl-paqtl-nt-v2",
        "amp-feng-20261007-eval-qtl-paqtl-hyenadna",
        "amp-feng-20261007-eval-qtl-paqtl-hyenadna-450k-long-sequence",
        "amp-feng-20261007-eval-qtl-paqtl-caduceus-ph",
        "amp-feng-20261007-eval-qtl-paqtl-caduceus-ph-long-sequence",
        "amp-feng-20261007-eval-qtl-paqtl-grover",
        "amp-feng-20261007-eval-qtl-paqtl-alphagenome-output-tracks"
      ],
      "endpoint": "paQTL causal-versus-noncausal variant discrimination (AUC, Cohen's d) for twelve DNA foundation-model/genomic-comparator configurations",
      "relevance": "proxy",
      "rationale": "paQTL causal-variant discrimination is a proxy for regulatory-variant gene follow-up prioritisation; it is not an experimentally validated follow-up choice and is a separate task from the other three QTL types.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Post-window-filter per-model scored denominators are unreported for paQTL; the pre-filter positive count (142) is not a confirmed denominator.",
        "AlphaGenome appears only in this QTL comparison, not the pathogenic/common SNP comparison. It uses the long-window QTL dataset, consuming central 131,072 bp and averaging output tracks over central 2,048 bp.",
        "Sei, Enformer (hidden/output tracks) and AlphaGenome (output tracks) are non-DNA-foundation specialised genomic comparators per the source's own asterisk notation, not DNA foundation models.",
        "No independent reproduction or qualified human scientific review."
      ],
      "citations": [
        {
          "source_id": "evidence-expansion-dna-foundation-models-2025-5d8ca9bc",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "c0275a28175c6f6dd8314d774021849363976dbef2f39b09da81dd8a3d8782ed"
    },
    {
      "id": "use-case-mapping-amp-20261007-feng-qtl-sqtl",
      "use_case_id": "use-case-regulatory-variant-gene-follow-up",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-feng-20261007-protocol-qtl-sqtl",
      "evaluation_ids": [
        "amp-feng-20261007-eval-qtl-sqtl-sei-hidden-states",
        "amp-feng-20261007-eval-qtl-sqtl-sei-output-tracks",
        "amp-feng-20261007-eval-qtl-sqtl-enformer-hidden-states",
        "amp-feng-20261007-eval-qtl-sqtl-enformer-output-tracks",
        "amp-feng-20261007-eval-qtl-sqtl-dnabert-2",
        "amp-feng-20261007-eval-qtl-sqtl-nt-v2",
        "amp-feng-20261007-eval-qtl-sqtl-hyenadna",
        "amp-feng-20261007-eval-qtl-sqtl-hyenadna-450k-long-sequence",
        "amp-feng-20261007-eval-qtl-sqtl-caduceus-ph",
        "amp-feng-20261007-eval-qtl-sqtl-caduceus-ph-long-sequence",
        "amp-feng-20261007-eval-qtl-sqtl-grover",
        "amp-feng-20261007-eval-qtl-sqtl-alphagenome-output-tracks"
      ],
      "endpoint": "sQTL causal-versus-noncausal variant discrimination (AUC, Cohen's d) for twelve DNA foundation-model/genomic-comparator configurations",
      "relevance": "proxy",
      "rationale": "sQTL causal-variant discrimination is a proxy for regulatory-variant gene follow-up prioritisation; it is not an experimentally validated follow-up choice and is a separate task from the other three QTL types.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Post-window-filter per-model scored denominators are unreported for sQTL; the pre-filter positive count (540) is not a confirmed denominator.",
        "AlphaGenome appears only in this QTL comparison, not the pathogenic/common SNP comparison. It uses the long-window QTL dataset, consuming central 131,072 bp and averaging output tracks over central 2,048 bp.",
        "Sei, Enformer (hidden/output tracks) and AlphaGenome (output tracks) are non-DNA-foundation specialised genomic comparators per the source's own asterisk notation, not DNA foundation models.",
        "No independent reproduction or qualified human scientific review."
      ],
      "citations": [
        {
          "source_id": "evidence-expansion-dna-foundation-models-2025-5d8ca9bc",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "84b40a538a5094d1faceaa17e4fd620b8713d5dff6b895b54eea4242769fcede"
    },
    {
      "id": "use-case-mapping-amp-20261007-feng-splice-acceptor",
      "use_case_id": "use-case-splicing-follow-up",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-feng-20261007-protocol-splice-acceptor",
      "evaluation_ids": [
        "amp-feng-20261007-eval-splice-acceptor-dnabert-2",
        "amp-feng-20261007-eval-splice-acceptor-nt-v2",
        "amp-feng-20261007-eval-splice-acceptor-hyenadna",
        "amp-feng-20261007-eval-splice-acceptor-caduceus-ph",
        "amp-feng-20261007-eval-splice-acceptor-grover"
      ],
      "endpoint": "Acceptor sequence-classification AUC for five DNA foundation-model embeddings",
      "relevance": "proxy",
      "rationale": "Sequence-label splice-acceptor classification AUC is a proxy for splicing-effect prioritisation; it is a different endpoint from MFASS exon-recognition reporter-assay ranking and must not be pooled with it or with the donor task.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Sequence-label acceptor classification is not a minigene reporter exon-inclusion endpoint and is not patient-RNA splicing validation.",
        "Exact per-task split sizes/counts are unextracted from the inspected main text (Supplementary Data 6 is identified but not inspected).",
        "No independent reproduction or qualified human scientific review."
      ],
      "citations": [
        {
          "source_id": "evidence-expansion-dna-foundation-models-2025-5d8ca9bc",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "c88a42cc23de2e92676eff53699c3b57d2f8d024886f3dbd620313d2d5a2fc29"
    },
    {
      "id": "use-case-mapping-amp-20261007-feng-splice-donor",
      "use_case_id": "use-case-splicing-follow-up",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-feng-20261007-protocol-splice-donor",
      "evaluation_ids": [
        "amp-feng-20261007-eval-splice-donor-dnabert-2",
        "amp-feng-20261007-eval-splice-donor-nt-v2",
        "amp-feng-20261007-eval-splice-donor-hyenadna",
        "amp-feng-20261007-eval-splice-donor-caduceus-ph",
        "amp-feng-20261007-eval-splice-donor-grover"
      ],
      "endpoint": "Donor sequence-classification AUC for five DNA foundation-model embeddings",
      "relevance": "proxy",
      "rationale": "Sequence-label splice-donor classification AUC is a proxy for splicing-effect prioritisation; it is a separate task from acceptor classification and from MFASS exon-recognition reporter-assay ranking, and must not be pooled with either.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Sequence-label donor classification is not a minigene reporter exon-inclusion endpoint and is not patient-RNA splicing validation.",
        "Exact per-task split sizes/counts are unextracted from the inspected main text (Supplementary Data 6 is identified but not inspected).",
        "No independent reproduction or qualified human scientific review."
      ],
      "citations": [
        {
          "source_id": "evidence-expansion-dna-foundation-models-2025-5d8ca9bc",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "7a805d2604c8eabefbdbd87b20be86922526da2270261c60bdb62307f139ae04"
    },
    {
      "id": "use-case-mapping-amp-20261007-issue10-lancet-virtual-tumor",
      "use_case_id": "use-case-tumour-dna-somatic-variant-detection",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-oncology-rna-20261007-protocol-lancet-virtual-tumor",
      "evaluation_ids": [
        "amp-oncology-rna-20261007-issue-10-eval-lancet",
        "amp-oncology-rna-20261007-issue-10-eval-strelka2"
      ],
      "endpoint": "Virtual-tumour SNV and indel precision/recall/F1, true/false positive and false negative counts for Lancet and Strelka2 on a real-read synthetic tumour/normal pair",
      "relevance": "direct",
      "rationale": "Real-read virtual-tumour spike-in directly measures somatic SNV/indel calling precision and recall, the declared benchmark endpoint, under one sequencing/normal regime.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Author-developed-caller evaluation, not independent reproduction.",
        "Virtual tumor endpoint does not establish sensitivity in clinical tumors, FFPE samples, panels, ctDNA, copy-number or structural variation.",
        "SNV printed counts agree with 31,592 truth variants. Do not repair MuTect table #calls=50,228 versus TP+FP=26,505; its row is not proposed for numerical ingestion.",
        "Indel Lancet and Strelka2 counts agree with 4,945 truth variants. Preserve Table 1 rounding and do not recompute percentages.",
        "Individual-condition confidence intervals unreported in Tables 1–2."
      ],
      "citations": [
        {
          "source_id": "amp-oncology-rna-20261007-source-pmc6123722",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "018e83af740b165027737c9b171ceb64ed144c548de9666ac12367c215027b78"
    },
    {
      "id": "use-case-mapping-amp-20261007-issue11-enfusion-nch-clinical",
      "use_case_id": "use-case-tumour-rna-fusion-detection",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-oncology-rna-20261007-protocol-enfusion-nch-clinical",
      "evaluation_ids": [
        "amp-oncology-rna-20261007-nch-eval-star-fusion",
        "amp-oncology-rna-20261007-nch-eval-arriba"
      ],
      "endpoint": "Retrospective sensitivity among 67 ensemble-ascertained clinically relevant fusions in a 229-sample paediatric cancer/haematologic cohort, for STAR-Fusion and Arriba",
      "relevance": "direct",
      "rationale": "A real clinical cohort with ensemble-ascertained clinically relevant fusions directly measures fusion-detection sensitivity in a clinical population, the declared endpoint; the ascertainment denominator is not an independent exhaustive truth set.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Ascertainment is from the optimized EnFusion pipeline; denominator is not an independent exhaustive truth set.",
        "Only a subset orthogonally confirmed; the endpoint does not quantify novel fusion generalization or patient outcomes.",
        "Do not combine with duplicate Seraseq measurements."
      ],
      "citations": [
        {
          "source_id": "amp-oncology-rna-20261007-source-pmc8642973",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "fa64eef53c2b9588535b6c09e919f81b7d6e8c9a0b19bf47cf5e211339dce30b"
    },
    {
      "id": "use-case-mapping-amp-20261007-issue11-enfusion-seraseq",
      "use_case_id": "use-case-tumour-rna-fusion-detection",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-oncology-rna-20261007-protocol-enfusion-seraseq",
      "evaluation_ids": [
        "amp-oncology-rna-20261007-issue-11-eval-arriba",
        "amp-oncology-rna-20261007-issue-11-eval-star-fusion",
        "amp-oncology-rna-20261007-issue-11-eval-enfusion-3-callers",
        "amp-oncology-rna-20261007-issue-11-eval-enfusion-3-callers-filter-known-fusion-list"
      ],
      "endpoint": "Synthetic 14-fusion Seraseq reference-standard sensitivity, precision and total/true fusion counts for Arriba, STAR-Fusion and EnFusion ensemble configurations (undiluted duplicate libraries)",
      "relevance": "direct",
      "rationale": "A synthetic reference standard with a known fusion set directly measures fusion-calling sensitivity and precision, the declared benchmark endpoint.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "The optimized known-fusion rescue uses prior knowledge of true reference fusions; 100% precision/sensitivity is an optimization-control observation, not unbiased generalization.",
        "Table2 caption v3 versus optimization narrative v2 is unresolved; dataset_version=null and automatic comparison blocked.",
        "STAR-Fusion precision 43.6% in Table2 vs 43.8% Par31; numeric candidate retains printed table value but requires disputed-result warning or omission pending resolution.",
        "All non-Seraseq fusions counted false positives although endogenous background fusions may exist.",
        "No CI or dispersion printed for Table2 rows.",
        "Clinical Table1 sensitivity is relative to 67 fusions ascertained through optimized ensemble, not independently exhaustive truth or treatment benefit."
      ],
      "citations": [
        {
          "source_id": "amp-oncology-rna-20261007-source-pmc8642973",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "46998424368e98a97385dea2498684ed51e5fa11037ea2666ed678ed62884995"
    },
    {
      "id": "use-case-mapping-amp-20261007-issue12-fraser-kremer",
      "use_case_id": "use-case-patient-rna-splicing-validation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-oncology-rna-20261007-protocol-fraser-kremer-subsampling",
      "evaluation_ids": [
        "amp-oncology-rna-20261007-issue-12-eval-fraser-2021-article-implementation"
      ],
      "endpoint": "Mean recovery of 13 known pathogenic splicing events at reduced cohort size (30 of 119 patient RNA samples) using FRASER on patient skin-fibroblast RNA",
      "relevance": "direct",
      "rationale": "Direct patient-RNA evidence measuring recovery of known pathogenic splicing events bears on the declared decision, scoped narrowly to known-event recovery rather than diagnostic yield in unknown cases.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Direct patient-RNA known-event recovery does not establish prospective diagnostic accuracy or patient outcome benefit.",
        "Known pathogenic-event ascertainment and same-cohort retrospective selection can bias recovery.",
        "30 is RNA sample subset size; do not silently equate to 30 independent patients.",
        "No blinded pathogenicity adjudication, general tissue transfer, VUS reclassification sensitivity or transcriptome-wide precision established by this endpoint.",
        "Subsampling repeats, exact score aggregation and numeric uncertainty unavailable from inspected main-text endpoint; supplementary PDF DNS retrieval failed.",
        "FRASER author correction 2022 changes GTEx version V7 to V6p; patient-Kremer endpoint not affected according to correction search result; exact correction bytes not retrieved."
      ],
      "citations": [
        {
          "source_id": "amp-oncology-rna-20261007-source-pmc7822922",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "bf9ed2b271943fe5d0a2622184d310ada3400f2318ed1728280fda8feaffa630"
    },
    {
      "id": "use-case-mapping-amp-20261007-issue13",
      "use_case_id": "use-case-diagnostic-dna-pathogen-identification",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-20261007-dna-pathogens-protocol",
      "evaluation_ids": [
        "amp-20261007-dna-pathogens-evaluation"
      ],
      "endpoint": "Positive percent agreement with initial blood culture for plasma microbial cfDNA sequencing in a prospective sepsis-alert cohort (59/63 initial-blood-culture-positive subset)",
      "relevance": "proxy",
      "rationale": "Positive percent agreement against initial blood culture is a proxy for sensitivity against all true infections, since initial blood culture does not identify every infection.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Positive agreement uses initial blood culture as reference, which does not identify all infections. It is not sensitivity against all true infections.",
        "Exact executable/software/database revision is not established; no foundation model and no new execution.",
        "Composite-reference outcomes and estimates are separate endpoints; Table 2 negative composite CI54.8–70.0 conflicts with narrative55.2–70.4 and is excluded.",
        "DNA-only plasma microbial cfDNA assay; no RNA viral detection or RNA sensitivity implied.",
        "Primary article hosted on third-party rious168 mirror; DOI/title match publisher. Public redistribution licence absent; only compact factual table excerpt retained, original PDF retrieval hash preserved."
      ],
      "citations": [
        {
          "source_id": "amp-20261007-dna-pathogens-source",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "2f8f6df06e1984966961023a929212ba443fba549e5ea12cb8f5592004ce552b"
    },
    {
      "id": "use-case-mapping-amp-20261007-issue14",
      "use_case_id": "use-case-diagnostic-rna-pathogen-detection",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-20261007-rna-pathogens-protocol",
      "evaluation_ids": [
        "amp-20261007-rna-pathogens-evaluation"
      ],
      "endpoint": "Sensitivity against original clinical respiratory-virus-panel testing (103/110, 93.6%) for RNA mNGS on a residual-sample pre-DTCA mixed respiratory-target cohort, adenovirus transcripts included",
      "relevance": "proxy",
      "rationale": "Sensitivity against original clinical RVP testing (RNA extraction, DNase treatment, cDNA synthesis; mixed respiratory-virus target panel including transcriptionally detected adenovirus) is a proxy for the declared RNA-pathogen diagnostic-accuracy decision: the specimen prep is an RNA workflow, not a mixed DNA/RNA sample-prep comparison, and this is not an RNA-virus-only subgroup score. The conflicting composite PPA figure (98.7%, 110.5/113) is excluded.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "RNA-only specimen preparation supports RNA pathogen-detection workflow; tested target mix includes adenovirus, a DNA virus detected via transcription. Do not describe the 93.6% as a pure RNA-virus-only subgroup score.",
        "Multiple detected targets are weighted so each specimen contributes one observation; out-of-panel mNGS positive calls are not counted as false positives.",
        "Positive-specimen BAL/swab count disagreement in Results vs Methods remains unresolved; total 110 positives and 81 negatives agree.",
        "No confidence interval extracted for original sensitivity. After selective discrepancy adjudication, report PPA/NPA rather than sensitivity; DTCA measurements must remain separate.",
        "No foundation model or independent external replication; clinical residual-sample validation study supplies conventional baseline evidence.",
        "Post-DTCA printed PPA98.7% (110.5/113) is arithmetically inconsistent; no DTCA endpoint selected or corrected."
      ],
      "citations": [
        {
          "source_id": "amp-20261007-rna-pathogens-source",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "4424bab553c78aafbbcf4a181468877e47b00bfc9ab02554c879483ef5f392cc"
    },
    {
      "id": "use-case-mapping-amp-20261007-issue15",
      "use_case_id": "use-case-plasma-ctdna-fragmentomics",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-20261007-ctdna-fragmentomics-protocol",
      "evaluation_ids": [
        "amp-20261007-ctdna-fragmentomics-evaluation"
      ],
      "endpoint": "Sensitivity (73%, 152/208) at a reported 98% specificity for plasma cfDNA fragmentomics-based cancer detection, repeated 10-fold cross-validation (10 repeats)",
      "relevance": "proxy",
      "rationale": "A repeated 10-fold cross-validation (10 repeats) on one assembled cohort is a proxy for the broader plasma ctDNA fragmentomics detection decision and for this explicitly assessed assay endpoint; it is internal cross-validation, not a separately held-out validation cohort and not prospective screening validation.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "4 of 215 healthy individuals misclassified at source-labelled 98% specificity. Retain rounded printed specificity; do not recompute sensitivity.",
        "Internal repeated cross-validation is not independent prospective screening validation; clinically identified cancers and healthy comparators differ from intended screening population.",
        "DELFI composite features include CNAs/mtDNA, so this exact result is not pure fragmentation alone.",
        "cfDNA signal is not purified ctDNA, tumour fraction limit is unreported for this endpoint.",
        "Combined mutation+DELFI 115/126 (91%) is a different subset/configuration and is excluded."
      ],
      "citations": [
        {
          "source_id": "amp-20261007-ctdna-fragmentomics-source",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "cd4db57e178c0edf3c49a8cbf8877bc769168345ab18d7cee3bbd32ee831f892"
    },
    {
      "id": "use-case-mapping-amp-20261007-issue16",
      "use_case_id": "use-case-plasma-ctdna-methylation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-protocol-cfmethyl-randomsplit-detection",
      "evaluation_ids": [
        "amp-eval-cfmethyl-allstage-sensitivity"
      ],
      "endpoint": "All-stage cancer-detection sensitivity (95% CI) at a declared specificity for cfMethyl-Seq on a repeated random 25% test split",
      "relevance": "direct",
      "rationale": "A repeated-split cross-validated sensitivity/specificity figure at a declared specificity directly measures plasma cfDNA methylation cancer-detection performance, the declared endpoint.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "No foundation-model comparison",
        "No prospective screening benefit",
        "No independent reproduction"
      ],
      "citations": [
        {
          "source_id": "amp-source-cfmethyl",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        },
        {
          "source_id": "amp-source-cfmethyl-correction",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "a4557cbe344cb50f36f257abb924d6a0f31b765c965dd695fae27dd18cf9327b"
    },
    {
      "id": "use-case-mapping-amp-20261007-issue17",
      "use_case_id": "use-case-cnv-detection-characterisation",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-protocol-hg002-cnv-1-5kb",
      "evaluation_ids": [
        "amp-eval-dragen-cnv-sv-1-5kb-fscore"
      ],
      "endpoint": "F-score for 1-5 kb deletion detection (DRAGEN 4.2 CNV/SV benchmarking, HG002) against the GIAB SV truth set",
      "relevance": "direct",
      "rationale": "F-score for 1-5 kb deletion detection against a GIAB truth set directly measures the declared CNV-detection endpoint; it does not cover visualisation usability.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "Unreported per-bin counts",
        "No duplication/tumourCNV/clinical endpoint",
        "No visualisation usability",
        "CNVnator table/prose conflict",
        "No foundation-model applicability"
      ],
      "citations": [
        {
          "source_id": "amp-source-dragen",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        },
        {
          "source_id": "amp-source-dragen-supplementary-tables",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "98e91a04803fe81a2302567c8942d3ec69a99c86179c781c6f06b5f9ba0a476e"
    },
    {
      "id": "use-case-mapping-amp-20261007-issue18",
      "use_case_id": "use-case-diagnostic-genomics-model-execution",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Add Codex-checked primary-source protocol evidence from the bounded AMP intake (rewire.it#365).",
      "protocol_id": "amp-protocol-dragen-hg002-phase4-total-runtime",
      "evaluation_ids": [
        "amp-eval-dragen-hg002-phase4-runtime"
      ],
      "endpoint": "Total single-sample wall-clock runtime (seconds) for DRAGEN 4.2 Phase 4 processing of HG002, by hardware configuration",
      "relevance": "proxy",
      "rationale": "A conventional variant-calling pipeline's measured single-sample runtime is a proxy baseline for the throughput/resource constraints a foundation-model execution workflow would need to meet; it is not itself a model-execution benchmark.",
      "constraints": [
        "Inspect every linked evaluation's source locator and preserved conflicts before citing a result.",
        "Do not combine this mapping's evaluations with any other protocol's results."
      ],
      "limitations": [
        "No foundation-model inference benchmark",
        "No peak memory",
        "No clinical reporting turnaround",
        "No runtime repeatCI"
      ],
      "citations": [
        {
          "source_id": "amp-source-dragen",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        },
        {
          "source_id": "amp-source-dragen-supplementary-tables",
          "locator": "See clinical/claims.csv and clinical/sources.md in data/omics/use-case-coverage-amp-20261007/"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Claude Sonnet AMP-integration worker, bounded transcription of Codex-checked primary values; independently reviewed by Codex (workbench/amp-supervision/primary-review.md, integration-review-corrections.md)",
        "reviewed_at": "2026-10-07T13:38:59Z",
        "note": "Bounded primary-source transcription, independently Codex-checked. No new model execution, independent experimental reproduction, qualified human scientific review or clinical validation."
      },
      "evidence_sha256": "8992490eb9aba66f496e6f9d8950a8b85405062638ca44d2bfed0dbe80e2dd59"
    },
    {
      "id": "use-case-mapping-msms-formula-no-formula-r1",
      "use_case_id": "use-case-mass-spectrum-molecule-shortlisting",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "msalign-2026-table3-task-massspecgym-formula-split-no-formula-r-1",
      "evaluation_ids": [
        "msalign-2026-table3-evaluation-deepsets-massspecgym-formula-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-emb-cos-massspecgym-formula-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-ffn-massspecgym-formula-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-jestr-massspecgym-formula-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-msalign-massspecgym-formula-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-sail-massspecgym-formula-split-no-formula-r-1"
      ],
      "endpoint": "Recall@1 of the true molecular structure among 256 mass-matched candidate structures on the MassSpecGym formula split, with no molecular formula supplied.",
      "relevance": "proxy",
      "rationale": "These source-checked retrieval evaluations inform method selection under explicit candidate and split assumptions; they do not establish accuracy in a new assay or unrestricted chemical search.",
      "constraints": [
        "Table 3, MassSpecGym formula split only; Recall@1. Retain all six formula-free configurations from that panel.",
        "Candidate sets contain 256 PubChem structures selected with a 10 ppm neutral-mass tolerance. The true molecular identity is assumed to be in the set.",
        "The formula data split groups by chemical formula; it does not supply formula to the evaluated models. Formula-conditioned methods and filtering are outside this mapping.",
        "These are the source paper implementations and training settings. Exact original-paper or current-release equivalence is not established.",
        "Metric groups at different shortlist sizes share an evaluation setting and must not count as independent replications."
      ],
      "limitations": [
        "The correct structure must be in a 256-candidate mass-filtered pool. Scores do not transfer automatically to a larger database, a different pool-construction procedure or a missing true structure.",
        "MCES and formula splits impose different distribution shifts. The source authors favour formula splits for their deployment interpretation; that choice does not prove that formula-split performance represents every intended application.",
        "Reported methods are the paper implementations: DeepSets training was extended to 50 epochs. These are not interchangeable with original model-paper scores or current checkpoints.",
        "Exact test denominators, split-manifest hashes, checkpoint hashes and evaluator revisions remain unextracted or unreported in the catalogue. No uncertainty values or new independent reproduction are established.",
        "Candidate recall is not an authenticated metabolite identification, a clinical diagnosis or a calibrated probability. Human domain review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "evidence-official-760ab2fa8c396aeb796c",
          "locator": "Table 3, MassSpecGym formula split block, row R@1, formula-free columns FFN, DeepSets, JESTR, Emb-Cos, SAIL and MSAlign; Sections 4, 5.1 and 5.2"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "264c270dc14e4c28740837553d6fcbf899529f22d829f825f065a798a4dfe76a"
    },
    {
      "id": "use-case-mapping-msms-formula-no-formula-r20",
      "use_case_id": "use-case-mass-spectrum-molecule-shortlisting",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "msalign-2026-table3-task-massspecgym-formula-split-no-formula-r-20",
      "evaluation_ids": [
        "msalign-2026-table3-evaluation-deepsets-massspecgym-formula-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-emb-cos-massspecgym-formula-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-ffn-massspecgym-formula-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-jestr-massspecgym-formula-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-msalign-massspecgym-formula-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-sail-massspecgym-formula-split-no-formula-r-20"
      ],
      "endpoint": "Recall@20 of the true molecular structure among 256 mass-matched candidate structures on the MassSpecGym formula split, with no molecular formula supplied.",
      "relevance": "proxy",
      "rationale": "These source-checked retrieval evaluations inform method selection under explicit candidate and split assumptions; they do not establish accuracy in a new assay or unrestricted chemical search.",
      "constraints": [
        "Table 3, MassSpecGym formula split only; Recall@20. Retain all six formula-free configurations from that panel.",
        "Candidate sets contain 256 PubChem structures selected with a 10 ppm neutral-mass tolerance. The true molecular identity is assumed to be in the set.",
        "The formula data split groups by chemical formula; it does not supply formula to the evaluated models. Formula-conditioned methods and filtering are outside this mapping.",
        "These are the source paper implementations and training settings. Exact original-paper or current-release equivalence is not established.",
        "Metric groups at different shortlist sizes share an evaluation setting and must not count as independent replications."
      ],
      "limitations": [
        "The correct structure must be in a 256-candidate mass-filtered pool. Scores do not transfer automatically to a larger database, a different pool-construction procedure or a missing true structure.",
        "MCES and formula splits impose different distribution shifts. The source authors favour formula splits for their deployment interpretation; that choice does not prove that formula-split performance represents every intended application.",
        "Reported methods are the paper implementations: DeepSets training was extended to 50 epochs. These are not interchangeable with original model-paper scores or current checkpoints.",
        "Exact test denominators, split-manifest hashes, checkpoint hashes and evaluator revisions remain unextracted or unreported in the catalogue. No uncertainty values or new independent reproduction are established.",
        "Candidate recall is not an authenticated metabolite identification, a clinical diagnosis or a calibrated probability. Human domain review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "evidence-official-760ab2fa8c396aeb796c",
          "locator": "Table 3, MassSpecGym formula split block, row R@20, formula-free columns FFN, DeepSets, JESTR, Emb-Cos, SAIL and MSAlign; Sections 4, 5.1 and 5.2"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "7b93e2000caca7115c4b176f38059899025b4efeb98f22a3e469bdc0c9521e1d"
    },
    {
      "id": "use-case-mapping-msms-mces-no-formula-r1",
      "use_case_id": "use-case-mass-spectrum-molecule-shortlisting",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "msalign-2026-table3-task-massspecgym-mces-split-no-formula-r-1",
      "evaluation_ids": [
        "msalign-2026-table3-evaluation-deepsets-massspecgym-mces-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-emb-cos-massspecgym-mces-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-ffn-massspecgym-mces-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-jestr-massspecgym-mces-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-msalign-massspecgym-mces-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-sail-massspecgym-mces-split-no-formula-r-1"
      ],
      "endpoint": "Recall@1 of the true molecular structure among 256 mass-matched candidate structures on the MassSpecGym MCES split, with no molecular formula supplied.",
      "relevance": "proxy",
      "rationale": "These source-checked retrieval evaluations inform method selection under explicit candidate and split assumptions; they do not establish accuracy in a new assay or unrestricted chemical search.",
      "constraints": [
        "Table 3, MassSpecGym MCES split only; Recall@1. Retain all six formula-free configurations from that panel.",
        "Candidate sets contain 256 PubChem structures selected with a 10 ppm neutral-mass tolerance. The true molecular identity is assumed to be in the set.",
        "The MCES split separates structure clusters with a minimum MCES distance greater than 10. Molecular formula is not supplied to these models; formula-conditioned methods and filtering are outside this mapping.",
        "These are the source paper implementations and training settings. Exact original-paper or current-release equivalence is not established.",
        "Metric groups at different shortlist sizes share an evaluation setting and must not count as independent replications."
      ],
      "limitations": [
        "The correct structure must be in a 256-candidate mass-filtered pool. Scores do not transfer automatically to a larger database, a different pool-construction procedure or a missing true structure.",
        "MCES and formula splits impose different distribution shifts. The source authors favour formula splits for their deployment interpretation; that choice does not prove that formula-split performance represents every intended application.",
        "Reported methods are the paper implementations: DeepSets training was extended to 50 epochs. These are not interchangeable with original model-paper scores or current checkpoints.",
        "Exact test denominators, split-manifest hashes, checkpoint hashes and evaluator revisions remain unextracted or unreported in the catalogue. No uncertainty values or new independent reproduction are established.",
        "Candidate recall is not an authenticated metabolite identification, a clinical diagnosis or a calibrated probability. Human domain review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "evidence-official-760ab2fa8c396aeb796c",
          "locator": "Table 3, MassSpecGym MCES split block, row R@1, formula-free columns FFN, DeepSets, JESTR, Emb-Cos, SAIL and MSAlign; Sections 4, 5.1 and 5.2"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "1556c455cef29ff84665800c8a0bfd3eacef978dcfb6f0048ed261d55e4bd55c"
    },
    {
      "id": "use-case-mapping-msms-mces-no-formula-r20",
      "use_case_id": "use-case-mass-spectrum-molecule-shortlisting",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "msalign-2026-table3-task-massspecgym-mces-split-no-formula-r-20",
      "evaluation_ids": [
        "msalign-2026-table3-evaluation-deepsets-massspecgym-mces-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-emb-cos-massspecgym-mces-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-ffn-massspecgym-mces-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-jestr-massspecgym-mces-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-msalign-massspecgym-mces-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-sail-massspecgym-mces-split-no-formula-r-20"
      ],
      "endpoint": "Recall@20 of the true molecular structure among 256 mass-matched candidate structures on the MassSpecGym MCES split, with no molecular formula supplied.",
      "relevance": "proxy",
      "rationale": "These source-checked retrieval evaluations inform method selection under explicit candidate and split assumptions; they do not establish accuracy in a new assay or unrestricted chemical search.",
      "constraints": [
        "Table 3, MassSpecGym MCES split only; Recall@20. Retain all six formula-free configurations from that panel.",
        "Candidate sets contain 256 PubChem structures selected with a 10 ppm neutral-mass tolerance. The true molecular identity is assumed to be in the set.",
        "The MCES split separates structure clusters with a minimum MCES distance greater than 10. Molecular formula is not supplied to these models; formula-conditioned methods and filtering are outside this mapping.",
        "These are the source paper implementations and training settings. Exact original-paper or current-release equivalence is not established.",
        "Metric groups at different shortlist sizes share an evaluation setting and must not count as independent replications."
      ],
      "limitations": [
        "The correct structure must be in a 256-candidate mass-filtered pool. Scores do not transfer automatically to a larger database, a different pool-construction procedure or a missing true structure.",
        "MCES and formula splits impose different distribution shifts. The source authors favour formula splits for their deployment interpretation; that choice does not prove that formula-split performance represents every intended application.",
        "Reported methods are the paper implementations: DeepSets training was extended to 50 epochs. These are not interchangeable with original model-paper scores or current checkpoints.",
        "Exact test denominators, split-manifest hashes, checkpoint hashes and evaluator revisions remain unextracted or unreported in the catalogue. No uncertainty values or new independent reproduction are established.",
        "Candidate recall is not an authenticated metabolite identification, a clinical diagnosis or a calibrated probability. Human domain review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "evidence-official-760ab2fa8c396aeb796c",
          "locator": "Table 3, MassSpecGym MCES split block, row R@20, formula-free columns FFN, DeepSets, JESTR, Emb-Cos, SAIL and MSAlign; Sections 4, 5.1 and 5.2"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "20b4477c9d7da7d6213c216c7729530cbc41539830e44375d8be88fd41a66c32"
    },
    {
      "id": "use-case-mapping-plant-promoter-maize-protoplasts-a-thaliana",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-maize-protoplasts-a-thaliana",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-maize-protoplasts-maize-protoplasts-a-thaliana",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-maize-protoplasts-maize-protoplasts-a-thaliana"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Arabidopsis thaliana sequences in maize protoplasts, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Arabidopsis thaliana promoter reporter experiments in maize protoplasts. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Arabidopsis thaliana, measured in the maize protoplasts reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 2 (data row 1), column R2; Species=A. thaliana; Model=Maize model; Type=AgroNT; Figures/Fig3_panele.txt, line 8 (data row 7), column R2; Species=A. thaliana; Model=Maize model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "51291aa4704085356887b47aa8c1326b2527631d19b94aaa9422bd7aae65c73d"
    },
    {
      "id": "use-case-mapping-plant-promoter-maize-protoplasts-s-bicolor",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-maize-protoplasts-s-bicolor",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-maize-protoplasts-maize-protoplasts-s-bicolor",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-maize-protoplasts-maize-protoplasts-s-bicolor"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Sorghum bicolor sequences in maize protoplasts, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Sorghum bicolor promoter reporter experiments in maize protoplasts. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Sorghum bicolor, measured in the maize protoplasts reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 3 (data row 2), column R2; Species=S. bicolor; Model=Maize model; Type=AgroNT; Figures/Fig3_panele.txt, line 9 (data row 8), column R2; Species=S. bicolor; Model=Maize model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "5471e1de9dd6577dcea6493aa23ff5afc73249f06b6e80e63f0a4727db673254"
    },
    {
      "id": "use-case-mapping-plant-promoter-maize-protoplasts-z-mays",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-maize-protoplasts-z-mays",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-maize-protoplasts-maize-protoplasts-z-mays",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-maize-protoplasts-maize-protoplasts-z-mays"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Zea mays sequences in maize protoplasts, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Zea mays promoter reporter experiments in maize protoplasts. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Zea mays, measured in the maize protoplasts reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 4 (data row 3), column R2; Species=Z. mays; Model=Maize model; Type=AgroNT; Figures/Fig3_panele.txt, line 10 (data row 9), column R2; Species=Z. mays; Model=Maize model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "475520831178904bc0c433ecd74d16b6336f195ce222cee5a249c18d6a36c6bc"
    },
    {
      "id": "use-case-mapping-plant-promoter-tobacco-leaves-a-thaliana",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-tobacco-leaves-a-thaliana",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-tobacco-leaves-tobacco-leaves-a-thaliana",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-tobacco-leaves-tobacco-leaves-a-thaliana"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Arabidopsis thaliana sequences in tobacco leaves, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Arabidopsis thaliana promoter reporter experiments in tobacco leaves. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Arabidopsis thaliana, measured in the tobacco leaves reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 5 (data row 4), column R2; Species=A. thaliana; Model=Tobacco model; Type=AgroNT; Figures/Fig3_panele.txt, line 11 (data row 10), column R2; Species=A. thaliana; Model=Tobacco model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "0ff8f7e2efa02a37a5ea958b65bb39cecb4c3bd5ede6320efe03cb5ba6b7ab81"
    },
    {
      "id": "use-case-mapping-plant-promoter-tobacco-leaves-s-bicolor",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-tobacco-leaves-s-bicolor",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-tobacco-leaves-tobacco-leaves-s-bicolor",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-tobacco-leaves-tobacco-leaves-s-bicolor"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Sorghum bicolor sequences in tobacco leaves, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Sorghum bicolor promoter reporter experiments in tobacco leaves. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Sorghum bicolor, measured in the tobacco leaves reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 6 (data row 5), column R2; Species=S. bicolor; Model=Tobacco model; Type=AgroNT; Figures/Fig3_panele.txt, line 12 (data row 11), column R2; Species=S. bicolor; Model=Tobacco model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "133865d16861493f70b2c49fd1c97a40a92964a2ceb2358a8d22cc7b72838611"
    },
    {
      "id": "use-case-mapping-plant-promoter-tobacco-leaves-z-mays",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-tobacco-leaves-z-mays",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-tobacco-leaves-tobacco-leaves-z-mays",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-tobacco-leaves-tobacco-leaves-z-mays"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Zea mays sequences in tobacco leaves, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Zea mays promoter reporter experiments in tobacco leaves. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Zea mays, measured in the tobacco leaves reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 7 (data row 6), column R2; Species=Z. mays; Model=Tobacco model; Type=AgroNT; Figures/Fig3_panele.txt, line 13 (data row 12), column R2; Species=Z. mays; Model=Tobacco model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "2ae000a01a35c24d345e32e99953aebcc54e266459abadebd9875174f75804d6"
    },
    {
      "id": "use-case-mapping-protein-stability-amfr-esm2",
      "use_case_id": "use-case-protein-stability",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial bounded applicability review of the existing AMFR ESM-2 evaluation.",
      "protocol_id": "rewire-protocol-proteingym-amfr-v13",
      "evaluation_ids": [
        "rewire-local-20260920-evaluation-proteingym-esm2"
      ],
      "endpoint": "Ranking the mixed AMFR assay cohort by proteolysis-inferred folding stability using ESM-2 8M masked marginals.",
      "relevance": "proxy",
      "rationale": "This completed short-construct assay is a narrow example for assessing stability-ranking evidence. Its mixed single/double cohort does not directly answer the planned single-substitution comparison or establish transfer to another protein.",
      "constraints": [
        "One 47-residue AMFR construct and all 2,972 variants, comprising 820 singles and 2,152 doubles; no train/test split.",
        "Exact esm2_t6_8M_UR50D checkpoint, wild-type-context masked marginals summed over substituted sites, CPU, one thread; no assay labels, alignment or structure as scorer inputs.",
        "Protocol-only mapping: a direct reviewed task-membership relationship is not recorded."
      ],
      "limitations": [
        "No interval or seed-variability estimate; training overlap remains unknown and assay bytes have not been independently authenticated against an upstream checksum.",
        "Recorded inference timing excludes loading, preparation and metrics; peak memory is unreported.",
        "The site-independent and full EVmutation methods in the planned singles study have no completed comparison here.",
        "No general whole-protein, ProteinGym-wide or clinical inference; automated source review is not human scientific validation."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260920-source-proteingym-esm2",
          "locator": "/metrics; /coverage; /execution/adapter_provenance; /execution; /input_information; /provenance; /protocol_results"
        },
        {
          "source_id": "rewire-local-20260920-instructions-proteingym",
          "locator": "Data and method; Reproduce"
        },
        {
          "source_id": "profile-protocol-proteingym-reference-files-dms-substitutions-csv-a8f49801",
          "locator": "AMFR_HUMAN_Tsuboyama_2023_4G3O row: seq_len, selection_assay, raw_DMS_phenotype_name and mutant-count columns"
        },
        {
          "source_id": "use-case-source-amfr-pilot-plan-194a78b",
          "locator": "Question; Task and data; Methods; Gates and execution order; Runtime and cost"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-25T15:28:11Z",
        "note": "Reviewed existing measured evidence separately from the planned singles comparison. No new experiment, human domain review or clinical validation."
      },
      "evidence_sha256": "81883d91b243e8a9c958bca976ad951aa82ba5a130eff9263dc8f3a04707012e"
    },
    {
      "id": "use-case-mapping-protein-stability-amfr-random",
      "use_case_id": "use-case-protein-stability",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial review of the separate fixed-seed random control as limited context.",
      "protocol_id": "rewire-protocol-proteingym-amfr-random-v13",
      "evaluation_ids": [
        "rewire-local-20260921-evaluation-proteingym-random"
      ],
      "endpoint": "One fixed-seed random ranking of the mixed AMFR stability cohort.",
      "relevance": "proxy",
      "rationale": "A recorded null control helps inspect the assay and evaluation procedure. It is not a biological prediction method recommendation, a chance-performance interval or a matched comparison with the separately executed ESM-2 protocol.",
      "constraints": [
        "All 2,972 AMFR variants, mixing single and double substitutions; a single fixed seed of 0.",
        "SHA256 ranking of prepared variant IDs; no biological prediction or label fitting.",
        "Protocol-only mapping; keep its evaluation and configuration separate from ESM-2 and the planned singles comparison."
      ],
      "limitations": [
        "One seed does not estimate chance variability or uncertainty. Do not subtract these separate protocol results to assert an evaluated winner.",
        "Local input hashes do not independently authenticate assay bytes against an upstream published checksum.",
        "This control supplies no pathogenicity or clinical suitability evidence; no human scientific review or independent replication is recorded."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260921-source-proteingym-random",
          "locator": "/metrics; /coverage; /model_configuration; /protocol_configuration; /provenance; /protocol_results"
        },
        {
          "source_id": "rewire-local-20260921-instructions-proteingym-random",
          "locator": "Pinned reproduction instructions and execution scope"
        },
        {
          "source_id": "use-case-source-amfr-pilot-plan-194a78b",
          "locator": "Question; Task and data; Methods (N0); Analysis populations"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-25T15:28:11Z",
        "note": "Reviewed as a separate one-seed control, not clinical evidence or a cross-protocol comparison."
      },
      "evidence_sha256": "2fa413430c25c90a4184be20e096c2200185e566d1690953e29bc40d55c98ddf"
    },
    {
      "id": "use-case-mapping-rhodopsin-wavelength-rhomax-fixed-probes",
      "use_case_id": "use-case-rhodopsin-wavelength-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "rewire-protocol-flip2-rhomax-by-wild-type-v1",
      "evaluation_ids": [
        "rewire-local-20260920-evaluation-flip2-composition",
        "rewire-local-20260920-evaluation-flip2-train-mean",
        "rewire-local-20260921-evaluation-composition22",
        "rewire-local-20260921-evaluation-esm2-35m",
        "rewire-local-20260921-evaluation-esm2-8m"
      ],
      "endpoint": "Spearman rank association and full-ranking NDCG for absorption-wavelength predictions on 184 held-out Rhomax sequences.",
      "relevance": "proxy",
      "rationale": "The archived background split directly tests one transfer setting; choosing constructs in a new experiment requires validation beyond those 184 rows.",
      "constraints": [
        "The same frozen Rhomax by_wild_type assignments cover 584 training, 116 validation and 184 test records. Validation labels do not select these fixed settings.",
        "Two exact ESM-2 checkpoints use frozen final-layer residue means, no MSA or templates, and a train-only alpha-10 ridge probe with train-only target scaling. They are not zero-shot likelihood scorers.",
        "The 40-feature and 22-feature composition controls are distinct configurations. Training mean is the existing constant-control evaluation; repeated control checks do not add replications."
      ],
      "limitations": [
        "A spectral-tuning endpoint is not opsin activation efficiency, expression, photostability, cellular function, general protein fitness or clinical usefulness.",
        "Spearman and full-ranking NDCG do not establish wavelength calibration, top-k precision or a prospective experimental hit rate. Constant predictions have undefined Spearman; their NDCG is a control value, not strong predictive evidence.",
        "No intervals or seed-variability estimates. The 35M checkpoint was chosen after seeing the 8M outcome; the source describes an exploratory follow-up, not a preregistered family-scale comparison.",
        "ESM-2 pretraining overlap is unresolved. A frozen linear probe result does not establish the value of alternative pooling, fine-tuning or every model-family configuration.",
        "Recorded timing covers sections of execution and is not a hardware-normalised deployment-cost comparison. Human domain review and external replication are not established."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260921-instructions-esm2-35m",
          "locator": "Matched Rhomax evaluations; Procedure and evidence; Re-run or inspect"
        },
        {
          "source_id": "rewire-local-20260921-source-esm2-35m",
          "locator": "/metrics; /coverage; /model_configuration; /protocol_configuration; /execution; /provenance"
        },
        {
          "source_id": "rewire-local-20260921-source-esm2-8m",
          "locator": "/metrics; /coverage; /model_configuration; /protocol_configuration; /execution; /provenance"
        },
        {
          "source_id": "rewire-local-20260921-source-composition22",
          "locator": "/metrics; /model_configuration; /protocol_configuration"
        },
        {
          "source_id": "rewire-local-20260920-source-flip2-train-mean",
          "locator": "/metrics; /coverage; /model_configuration"
        },
        {
          "source_id": "rewire-local-20260920-source-flip2-composition",
          "locator": "/metrics; /coverage; /model_configuration; /provenance"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "a012ef5867a510fe9861d1a3de019095e38fb8718f0dbd3581ddd1c5b9a51783"
    },
    {
      "id": "use-case-mapping-splicing-mfass-matched-v1",
      "use_case_id": "use-case-splicing-follow-up",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial bounded applicability review of the matched MFASS study.",
      "protocol_id": "rewire-mfass-matched-v1-protocol",
      "task_id": "catalog-task-mfass-splice",
      "evaluation_ids": [
        "rewire-mfass-matched-v1-evaluation-s0",
        "rewire-mfass-matched-v1-evaluation-s1",
        "rewire-mfass-matched-v1-evaluation-p0",
        "rewire-mfass-matched-v1-evaluation-p1"
      ],
      "endpoint": "Ranking held-out MFASS SNVs by reporter-assay splice disruption under matched canonical annotation.",
      "relevance": "proxy",
      "rationale": "The recorded endpoint informs assay-oriented prioritisation under its declared conditions. Selecting variants for a different follow-up experiment requires transfer validation; the endpoint is not patient RNA or clinical pathogenicity.",
      "constraints": [
        "Identical scored population: 8,297 of 8,324 held-out variants, 314 scored positives and 460 groups in all four configurations.",
        "Shared GENCODE 44 canonical transcript selection and FASTA; 50-base distance; distinct SpliceAI and Pangolin masking settings.",
        "Keep historical annotation conditions and scoring populations in their existing separate comparison groups."
      ],
      "limitations": [
        "23 assembly-orientation mismatches and four canonical-transcript-span exclusions remain unscored. The latter are protocol exclusions, not established faulty variants; full-population performance is unknown.",
        "No top-100 precision difference is established and masked Pangolin is tie-sensitive. Paired contrast intervals are not individual-condition intervals.",
        "Exploratory source review only; no independent human review, replication or clinical validation. Assembly-issue author confirmation is not established."
      ],
      "citations": [
        {
          "source_id": "rewire-mfass-matched-v1-source-report",
          "locator": "/conditions/{S0,S1,P0,P1}/coverage; /conditions/{S0,S1,P0,P1}/metrics; /conditions/{S0,S1,P0,P1}/ties; /contrasts/{S1-S0,P1-P0,P0-S0}/paired/precision_at_capacity"
        },
        {
          "source_id": "rewire-mfass-matched-v1-source-manifest-v1",
          "locator": "/conditions; /distance; /resources_sha256; /code"
        },
        {
          "source_id": "rewire-mfass-matched-v1-source-exclusion-verification",
          "locator": "/exclusion_counts; /conditions; /checks"
        },
        {
          "source_id": "use-case-source-mfass-matched-intake-194a78b",
          "locator": "Opening coverage, exclusion, annotation and interpretation paragraphs; Review and validation"
        },
        {
          "source_id": "evidence-expansion-mfass-readme-62a93814",
          "locator": "Dataset; Limits"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-25T15:28:11Z",
        "note": "Reviewed exact configuration and population scope against pinned source bytes. Applicability remains proxy evidence; no human domain review or clinical validation."
      },
      "evidence_sha256": "c4cfbe66c30d59308fc8d2226f42720a09a6e54d2f407c7141b08e01e39c0a47"
    },
    {
      "id": "use-case-mapping-utr-translation-baselines-designed-v1",
      "use_case_id": "use-case-utr-translation-baselines",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "rewire-protocol-mrnabench-designed-mrl-v1",
      "evaluation_ids": [
        "rewire-local-20260920-evaluation-mrnabench-composition",
        "rewire-local-20260920-evaluation-mrnabench-train-mean"
      ],
      "endpoint": "Predicting target_mrl_designed on the complete 15,003-record canonical test split of the Sample designed reporter dataset.",
      "relevance": "direct",
      "rationale": "The matched sequence-composition and training-mean controls provide an auditable starting point for deciding whether a more complex method adds predictive value on this defined endpoint. They do not establish transfer to a different library or supply a pretrained-model comparison.",
      "constraints": [
        "Both controls use the same full processed sequences, seed-2541 split, training labels and complete held-out test population. No validation or test labels select hyperparameters.",
        "Composition features are log1p length, A/C/G/T fractions and unknown fraction, fitted with train-only RidgeCV over the recorded five alpha values and no normalization. The constant control is the training-target arithmetic mean.",
        "MSE is the matched primary endpoint. Constant-control Pearson and Spearman are unavailable; do not replace missing values with zeros."
      ],
      "limitations": [
        "One target, one split and one execution per method; no uncertainty or seed variability. No homology-separated or compositional-generalisation claim.",
        "This training-only protocol differs from upstream probing and is not an aggregate mRNABench score or a reproduction of a published model result.",
        "The original run’s raw inputs and saved predictions are not publicly archived. The recipe obtains source data and runs the controls again, including separate protein controls; it does not rescore prior predictions. Source-data reuse terms and cross-platform reproducibility remain unresolved.",
        "Research baseline evidence only, with no new execution, human domain review or clinical validation."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260920-source-mrnabench-composition",
          "locator": "/coverage; /model_configuration; /protocol_configuration; /protocol_results; /provenance"
        },
        {
          "source_id": "rewire-local-20260920-source-mrnabench-train-mean",
          "locator": "/coverage; /model_configuration; /protocol_results/metric_unavailable_reasons; /provenance"
        },
        {
          "source_id": "rewire-local-20260920-instructions-mrnabench",
          "locator": "Results: RNA evaluation paragraph; Provenance and review; Reproduce the four sequence controls"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:15:35.948Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "1034171098b5ead5f29764df1c6efaccee4aa4c24f235945e36b8536ec30caf5"
    }
  ],
  "release_id": "2026-10-07-1448159e6a81",
  "input_sha256": "d60fd7f669bfec7bd34ec6e5080d8e1cb4e8186f8286ead60888848f1e20e001"
}
