{
  "schema_version": "1.0",
  "use_cases": [
    {
      "id": "use-case-brca1-brca2-germline-interpretation",
      "slug": "brca1-brca2-germline-interpretation",
      "title": "Interpret BRCA1/BRCA2 germline variants",
      "question": "Which evidence-support methods improve BRCA1/BRCA2 variant review without increasing serious classification errors?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "hereditary cancer",
        "germline",
        "BRCA1",
        "BRCA2",
        "ENIGMA",
        "variant classification",
        "VUS",
        "ACMG AMP"
      ],
      "intended_users": [
        "Germline variant scientists",
        "Hereditary-cancer genetics reviewers"
      ],
      "decision": "Choose methods that assemble and assess variant evidence under gene-specific criteria for a professional BRCA1/BRCA2 classification review.",
      "inputs": [
        "Confirmed germline variant, reference transcript and assay limitations",
        "Population frequency, segregation, phenotype and functional evidence with dates",
        "A relevant, versioned ClinGen ENIGMA specification and evidence cutoff"
      ],
      "output": "A classification dossier with criterion-level evidence, conflicts, uncertainty and further information needed for review.",
      "setting": "Germline BRCA1/BRCA2 interpretation in hereditary-cancer testing. Additional genes require their own specifications and comparisons.",
      "exclusions": [
        "Germline classification does not estimate an individual's absolute cancer risk or select treatment.",
        "Tumour-only sequencing does not confirm germline status.",
        "Functional scores and computational predictions are evidence inputs, not complete classifications."
      ],
      "clinical_scope": "Clinical research on hereditary-cancer evidence review. A VUS does not justify management changes; a negative panel does not remove risk associated with family history.",
      "evidence_gaps": [
        "Independent classifications with documented criteria, date and predictor contributions remain to be collected.",
        "Qualified hereditary-cancer reviewers are needed to adjudicate serious errors and unresolved or conflicting evidence.",
        "Public assertions may incorporate the method being tested and cannot be assumed to provide independent labels."
      ],
      "citations": [
        {
          "source_id": "use-case-source-clinical-priorities-2026-09-28",
          "locator": "C3 — BRCA1/BRCA2 germline interpretation"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex",
        "reviewed_at": "2026-09-28T14:52:02.000Z",
        "note": "Automated review of Rewire's workflow definition and sourced research brief against the agreed exploration and backlog. No human expert approval, model-applicability assessment or clinical validation is claimed."
      },
      "collection_plan": {
        "status": "planned",
        "comparison_question": "Can an evidence-support method reduce review effort without increasing serious classification or criteria errors relative to standard gene-specific review?",
        "baselines": [
          "Manual criteria-based review using the same evidence cutoff and ClinGen ENIGMA specification"
        ],
        "outcomes": [
          "Serious classification disagreements and criterion-assignment errors",
          "Coverage of unresolved cases and preservation of conflicts",
          "Review time against independent expert adjudication"
        ],
        "validation_requirements": [
          "Record the specification version, classification date and germline confirmation.",
          "Audit predictor circularity and overlap with functional training data.",
          "Use temporal and gene/domain separation appropriate to the intended claim.",
          "Retain uncertain and conflicting cases and predefine error severity.",
          "Obtain blinded independent adjudication by qualified hereditary-cancer reviewers."
        ],
        "next_step": "Prepare a BRCA1/BRCA2 interpretation evidence matrix with criteria versions, label provenance, error severity and unmet independent-adjudication needs."
      },
      "planned_work": []
    },
    {
      "id": "use-case-cell-type-annotation-transfer",
      "slug": "cell-type-annotation-transfer",
      "title": "Transfer cell-type annotations to a new dataset",
      "question": "Which annotation workflow can label my new dataset reliably and recognise unsupported cell populations?",
      "area": "cells-tissues",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "cell-type annotation",
        "reference transfer",
        "label transfer",
        "single-cell",
        "single-nucleus",
        "cell ontology",
        "rare cells",
        "unknown cell types",
        "Azimuth",
        "CellTypist",
        "scANVI",
        "scTab"
      ],
      "intended_users": [
        "Single-cell analysts",
        "Cell atlas researchers"
      ],
      "decision": "Choose an annotation workflow and reference, accept supported labels, and identify cells that require expert review or additional measurements.",
      "inputs": [
        "Query count data with tissue, disease, donor and assay metadata",
        "A versioned reference and a defined cell-label hierarchy",
        "Independent marker, protein or expert evidence where available",
        "Requirements for coverage, uncertainty and resource use"
      ],
      "output": "Cell-type labels at the requested ontology level, confidence estimates and an explicit unassigned group.",
      "setting": "Research annotation of a new single-cell or single-nucleus cohort. Transfer across donors, studies, disease contexts and technologies is assessed separately.",
      "exclusions": [
        "Atlas labels are not infallible ground truth.",
        "Embedding separation and batch mixing do not establish correct cell identity.",
        "Annotation, batch integration and discovery of new disease states require separate evaluations."
      ],
      "clinical_scope": "Research only. Annotation accuracy does not establish the validity of a clinical diagnostic classifier.",
      "evidence_gaps": [
        "Comparisons need independent study and platform holdouts with reference and pretraining provenance.",
        "Rare and absent-reference populations need explicit evaluation alongside common cell types.",
        "Human domain review remains unassigned."
      ],
      "citations": [
        {
          "source_id": "use-case-source-research-priorities-2026-09-28",
          "locator": "R4 — Cell-type annotation transfer"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex",
        "reviewed_at": "2026-09-28T14:52:13Z",
        "note": "Automated review of the workflow definition, cited primary-source scope and collection plan. No model evaluation or applicability mapping was added. Human domain review remains unassigned."
      },
      "collection_plan": {
        "status": "planned",
        "comparison_question": "Does a method improve known-type annotation and unknown-type recognition at matched coverage over conventional methods across independent studies?",
        "baselines": [
          "Marker-based and nearest-reference annotation",
          "Appropriate pinned Azimuth, CellTypist or scANVI workflows",
          "Embeddings with a simple classifier where applicable"
        ],
        "outcomes": [
          "Class-specific and hierarchy-aware errors",
          "Rare-type and absent-reference performance",
          "Calibration and accuracy as uncertain cells are left unassigned",
          "Resource use and transfer across studies and platforms"
        ],
        "validation_requirements": [
          "Hold out donors, whole studies and platforms as required by the intended query.",
          "Audit duplicated cells, donors and labels across references and pretraining data.",
          "Use independent expert or orthogonal label checks and preserve uncertainty in reference labels.",
          "Match ontology level and coverage; keep integration and disease-state discovery separate."
        ],
        "next_step": "Build an annotation-transfer evidence table with reference versions, donor/study independence, label hierarchy and coverage-aware outcomes."
      },
      "planned_work": []
    },
    {
      "id": "use-case-egfr-nsclc-actionability-resistance-evidence",
      "slug": "egfr-nsclc-actionability-resistance-evidence",
      "title": "Review EGFR lung-cancer actionability evidence",
      "question": "Which systems retrieve correctly scoped sensitivity and resistance evidence for advanced EGFR-mutant NSCLC?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "EGFR",
        "NSCLC",
        "non-small-cell lung cancer",
        "molecular tumour board",
        "actionability",
        "resistance",
        "CIViC",
        "evidence retrieval"
      ],
      "intended_users": [
        "Molecular tumour-board scientists",
        "Oncologists and clinical molecular scientists reviewing evidence"
      ],
      "decision": "Choose an evidence-review system that connects tumour findings to relevant sensitivity and resistance studies while exposing contradictions and missing context.",
      "inputs": [
        "Interpreted biomarker, genome build/transcript, histology, stage and co-alterations",
        "Prior therapies, progression and sampling dates, including tissue or plasma source",
        "Jurisdiction, evidence cutoff and a versioned source corpus"
      ],
      "output": "Source-linked evidence assertions with regimen, sensitivity or resistance direction, treatment setting, native evidence tier, contradictions and unresolved information.",
      "setting": "Advanced EGFR-mutant non-small-cell lung cancer, including progression after EGFR-targeted therapy. Treatment history and evidence date are part of every comparison.",
      "exclusions": [
        "Evidence retrieval does not select a treatment or establish individual treatment benefit.",
        "Trial eligibility, response and survival require separate evaluations.",
        "Missing evidence does not establish resistance, and evidence tiers from different systems are not interchangeable."
      ],
      "clinical_scope": "Clinical research on support for qualified molecular-oncology evidence review. Applicability depends on tumour, biomarker, prior therapy, sampling context, jurisdiction and evidence date.",
      "evidence_gaps": [
        "Independently adjudicated case/evidence sets and comparisons at equal review effort remain to be collected.",
        "Corrected guidance and primary studies need detailed context extraction and qualified molecular-oncology review.",
        "Some comparators require specific reuse permission; public papers may not provide reusable patient-level data."
      ],
      "citations": [
        {
          "source_id": "use-case-source-clinical-priorities-2026-09-28",
          "locator": "C5 — EGFR-mutant NSCLC actionability and resistance evidence"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex",
        "reviewed_at": "2026-09-28T14:52:02.000Z",
        "note": "Automated review of Rewire's workflow definition and sourced research brief against the agreed exploration and backlog. No human expert approval, model-applicability assessment or clinical validation is claimed."
      },
      "collection_plan": {
        "status": "planned",
        "comparison_question": "Does retrieval-assisted review recover more correctly scoped sensitivity/resistance evidence with fewer unsupported assertions than a versioned lookup/manual-search workflow at equal review effort?",
        "baselines": [
          "Versioned knowledgebase lookup plus conventional manual primary-source review"
        ],
        "outcomes": [
          "Relevant-evidence recall and unsupported-assertion rate",
          "Tumour, treatment-line, direction, date and citation fidelity",
          "Contradiction handling, missing-information handling and review time"
        ],
        "validation_requirements": [
          "Freeze the question, corpus and knowledge cutoff; hold out studies and molecular profiles.",
          "Preserve native evidence tiers and include contradictory or inapplicable evidence.",
          "Audit citation support and retain unknown clinical information.",
          "Use independent qualified molecular-oncology adjudication.",
          "Confirm source-specific benchmarking permissions and separate evidence review from treatment and eligibility outcomes."
        ],
        "next_step": "Prepare a versioned NSCLC evidence-review specification and extraction matrix covering biomarker, regimen, line, prior treatment, direction, source and date, with source-permission dependencies recorded separately."
      },
      "planned_work": []
    },
    {
      "id": "use-case-genetic-perturbation-response",
      "slug": "genetic-perturbation-response",
      "title": "Assess models for genetic perturbation experiments",
      "question": "Which prediction methods and controls should I test before using expression predictions to plan genetic perturbation experiments?",
      "area": "cells-tissues",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "genetic perturbation",
        "perturbation response",
        "Perturb-seq",
        "single-cell RNA-seq",
        "scRNA-seq",
        "GEARS",
        "CPA",
        "knowledge graph",
        "Norman2019",
        "no-change control",
        "gene expression"
      ],
      "intended_users": [
        "Researchers designing genetic perturbation screens",
        "Computational biologists evaluating perturbation-response models"
      ],
      "decision": "Choose models and controls for a pilot in the intended cell system before using predictions to select follow-up experiments. The current evidence compares expression prediction; it does not establish experimental hit rates.",
      "inputs": [
        "Single-cell expression measurements from perturbed and unperturbed cells, with perturbation and cell-type labels",
        "A study design with multiple measured perturbations and cells per perturbation; combinations require combination examples during GEARS training"
      ],
      "output": "Predicted post-perturbation expression and changes relative to unperturbed controls, assessed against measured responses.",
      "setting": "Research method assessment anchored to the Norman2019 K562 cell-line comparison in GEARS Supplementary Table 6. Both endpoints below describe the same four evaluated configurations.",
      "exclusions": [
        "Cross-cell-type transfer and bulk-sequencing prediction are outside the cited GEARS usage scope.",
        "Training GEARS only on single-gene perturbations does not support reliable combination prediction.",
        "Expression prediction scores do not establish causal mechanism, fitness effects or successful experimental prioritisation."
      ],
      "clinical_scope": "Research use only. These expression endpoints do not establish diagnostic, treatment-selection or patient-response validity.",
      "evidence_gaps": [
        "The exact Table 6 split manifest, scoring gene subset, scored counts and checkpoint revisions remain unextracted. Do not silently label the MSE endpoint as Top20.",
        "The table does not identify its printed spread as SD, SE or a confidence interval; these scores do not establish a statistically supported winner.",
        "Both endpoints and the repeated No Perturb row are overlapping evidence from one source comparison, not independent replications.",
        "Performance in a new cell system and experimental hit rates need a separate prospective evaluation. Human scientific review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "coverage-source-gears-supp",
          "locator": "Supplementary Table 6, printed page 34 / PDF page 35; Supplementary Table 1, PDF page 30; Supplementary Notes 5 and 14 (K562 context)."
        },
        {
          "source_id": "evidence-official-c037e3419c04936262a0",
          "locator": "README.md at f374e43e197b295016d80395d7a54ddb81cc6769, A note on usage (lines 15–19) and Core API Interface (lines 20–54)."
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "planned_work": []
    },
    {
      "id": "use-case-mass-spectrum-molecule-shortlisting",
      "slug": "mass-spectrum-molecule-shortlisting",
      "title": "Shortlist molecular identities from tandem mass spectra",
      "question": "Which retrieval configurations merit validation when molecular formula is unknown and a researcher can examine only a short list of candidate structures?",
      "area": "metabolomics",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "metabolomics",
        "tandem mass spectrometry",
        "MS/MS",
        "molecular identification",
        "candidate retrieval",
        "unknown formula",
        "MassSpecGym",
        "MSAlign",
        "DreaMS",
        "ChemBERTa",
        "MCES",
        "formula split"
      ],
      "intended_users": [
        "Metabolomics researchers shortlisting candidate molecular identities for experimental confirmation",
        "Computational researchers choosing retrieval inputs, candidate pools and validation splits"
      ],
      "decision": "Compare formula-free retrieval configurations within a named split and shortlist size, then validate the intended instrument, chemistry and candidate database before selecting a workflow.",
      "inputs": [
        "MS/MS spectra with metadata sufficient to derive neutral molecular mass",
        "A frozen set of candidate molecular structures; the true identity must be present for the measured retrieval endpoint"
      ],
      "output": "Split-specific Recall@1 and Recall@20 evidence for six exact configurations, plus candidate-pool and domain-shift limits.",
      "setting": "MSAlign paper Table 3, MassSpecGym formula and MCES splits, formula-free inputs. Each source candidate pool contains 256 mass-matched PubChem structures. Formula and MCES splits remain distinct.",
      "exclusions": [
        "De novo molecular generation or cases where the true structure is absent from the candidate set",
        "Formula-conditioned retrieval, including MIST, FLARE and MSAlign+Filter",
        "Pooling split-specific results or transferring scores to different candidate database sizes",
        "Authenticated molecular identification and clinical diagnosis"
      ],
      "clinical_scope": "Research only. Candidate retrieval is not a confirmed molecular identification or evidence of clinical diagnostic performance.",
      "evidence_gaps": [
        "The correct structure must be in a 256-candidate mass-filtered pool. Scores do not transfer automatically to a larger database, a different pool-construction procedure or a missing true structure.",
        "MCES and formula splits impose different distribution shifts. The source authors favour formula splits for their deployment interpretation; that choice does not prove that formula-split performance represents every intended application.",
        "Reported methods are the paper implementations: DeepSets training was extended to 50 epochs. These are not interchangeable with original model-paper scores or current checkpoints.",
        "Exact test denominators, split-manifest hashes, checkpoint hashes and evaluator revisions remain unextracted or unreported in the catalogue. No uncertainty values or new independent reproduction are established.",
        "Candidate recall is not an authenticated metabolite identification, a clinical diagnosis or a calibrated probability. Human domain review remains outstanding.",
        "The four metric and split groups come from one paper and do not constitute independent replications."
      ],
      "citations": [
        {
          "source_id": "evidence-official-760ab2fa8c396aeb796c",
          "locator": "Table 3, MassSpecGym MCES and formula blocks, formula-free columns, R@1 and R@20; Sections 4, 5.1 and 5.2"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "planned_work": []
    },
    {
      "id": "use-case-phenotype-perturbation-selection",
      "slug": "phenotype-perturbation-selection",
      "title": "Select perturbations for a defined cellular response",
      "question": "Which genetic perturbations should I test to produce a defined cellular response?",
      "area": "cells-tissues",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "perturbation selection",
        "phenotype",
        "genetic screen",
        "CRISPR",
        "Perturb-seq",
        "experimental hit rate",
        "cellular response",
        "unseen perturbation",
        "Virtual Cell Challenge",
        "Systema"
      ],
      "intended_users": [
        "Functional genomics screen designers",
        "Cell biologists planning perturbation experiments"
      ],
      "decision": "Allocate a fixed experimental budget to perturbations most likely to produce the prespecified phenotype.",
      "inputs": [
        "A defined cell system, starting state and target phenotype",
        "Candidate genetic perturbations and a fixed testing budget",
        "Measured perturbations and matched controls with guide, replicate and condition metadata",
        "The intended transfer setting: unseen genes, combinations or a new cellular context"
      ],
      "output": "A ranked experimental shortlist with predicted phenotype effects, coverage and uncertainty about unmeasured conditions.",
      "setting": "Research selection for a defined genetic-perturbation screen. Chemical interventions and combinations need their own protocols. This question concerns phenotype hits rather than expression reconstruction alone.",
      "exclusions": [
        "Transcriptome similarity alone does not establish successful experimental selection.",
        "Generalisation to an unseen gene, combination and cell context must be evaluated separately.",
        "Choosing experiments to maximise information gain is a separate decision."
      ],
      "clinical_scope": "Research only. Predicted cellular responses do not establish patient response or treatment-selection validity.",
      "evidence_gaps": [
        "Fixed-budget experimental hit-selection comparisons have not yet been collected for this question.",
        "Expression benchmarks may provide only intermediate evidence for the chosen phenotype.",
        "Human domain review remains unassigned."
      ],
      "citations": [
        {
          "source_id": "use-case-source-research-priorities-2026-09-28",
          "locator": "R3 — Phenotype-driven perturbation selection"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex",
        "reviewed_at": "2026-09-28T14:52:13Z",
        "note": "Automated review of the workflow definition, cited primary-source scope and collection plan. No model evaluation or applicability mapping was added. Human domain review remains unassigned."
      },
      "collection_plan": {
        "status": "planned",
        "comparison_question": "Does a model select more perturbations producing a prespecified phenotype than simple controls at the same test budget in the intended held-out setting?",
        "baselines": [
          "Random candidate selection and mean-response/no-change controls",
          "Linear or other simple statistical response models",
          "Nearest measured perturbation"
        ],
        "outcomes": [
          "Experimentally confirmed phenotype hits per test budget",
          "Perturbation-specific expression signal as an intermediate endpoint",
          "Coverage and uncertainty for unseen interventions or contexts"
        ],
        "validation_requirements": [
          "Use the same candidate universe, phenotype definition and budget; specify how tied rankings are sampled.",
          "Separate unseen genes, combinations and cell contexts, holding out biological and experimental units.",
          "Audit guide efficacy, shared controls, batch effects and model training overlap.",
          "Do not choose evaluation genes using held-out responses; keep information-gain design separate."
        ],
        "next_step": "Inventory perturbation studies with control and split metadata, separating confirmed phenotype-selection outcomes from expression-only metrics."
      },
      "planned_work": []
    },
    {
      "id": "use-case-plant-promoter-reporters",
      "slug": "plant-promoter-reporters",
      "title": "Compare methods for plant promoter experiments",
      "question": "What evidence supports choosing a sequence model for predicting plant core-promoter strength in the intended reporter assay?",
      "area": "dna-genomes",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "plant promoter",
        "core promoter",
        "promoter strength",
        "synthetic biology",
        "STARR-seq",
        "AgroNT",
        "CNN",
        "maize protoplasts",
        "tobacco leaves",
        "Arabidopsis thaliana",
        "Sorghum bicolor",
        "Zea mays"
      ],
      "intended_users": [
        "Plant synthetic-biology researchers prioritising promoter candidates for reporter experiments",
        "Computational researchers selecting an assay-matched promoter prediction method"
      ],
      "decision": "Choose the matching sequence species and reporter host, then inspect the two evaluated methods and remaining validation needs before prioritising new promoter candidates. Keep each assay and species comparison separate.",
      "inputs": [
        "A 170 bp core-promoter sequence spanning −165 to +5 relative to an annotated transcription start site",
        "The sequence species and intended assay host: maize protoplasts or tobacco leaves"
      ],
      "output": "Assay-specific evidence for AgroNT and the CNN from Jores et al., with exact evaluated configurations and source rows; no general plant-performance ranking.",
      "setting": "The evidence concerns held-out core-promoter sequences from Arabidopsis thaliana, Sorghum bicolor and Zea mays. Their activity was measured by transient STARR-seq reporter assays in maize protoplasts or tobacco leaves, using the original studies’ train/test datasets for the model comparisons.",
      "exclusions": [
        "Endogenous gene expression, stable transformed plants, tissue-specific activity outside the two reporter hosts, plant fitness or crop yield",
        "Terminator strength, promoter–terminator combinations, chromatin accessibility and other AgroNT tasks",
        "Prospective validation of newly designed promoters, or pooling the six assay-by-species conditions into one score"
      ],
      "clinical_scope": "This page concerns plant reporter research. The evidence establishes no clinical application.",
      "evidence_gaps": [
        "All values are author-reported and source checked; no independent experimental reproduction or human scientific review is recorded.",
        "The source tables provide no confidence intervals or run variability. Exact fitted checkpoint hashes, task-specific seeds, scoring denominators and executable split manifests remain unreported or unextracted.",
        "The six comparisons retain their own assay system and sequence species. A maize-protoplast model can be evaluated on sequences from any of the three listed species; the host is not the sequence origin.",
        "Held-out labelled examples do not establish exclusion from AgroNT pretraining or generalisation to an unseen species. Its reference-genome pretraining and supervised assay split answer different leakage questions.",
        "Held-out R² describes assay-strength prediction. It is not top-k candidate hit rate, uncertainty for a new promoter, or proof that a designed promoter will work in a stable plant.",
        "These four task-specific configurations do not establish performance for other AgroNT checkpoints or newer methods. Missing checkpoint and fitting detail limits exact recreation of the reported comparison.",
        "The separately flagged Figure 4 chromatin-accessibility source anomaly is outside this page and supplies no evidence for promoter selection."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, lines 2–13, R2 column; pair rows by Species, Model and Type"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21); pretraining data and training (Sec14–15)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "planned_work": []
    },
    {
      "id": "use-case-protein-stability",
      "slug": "protein-variant-stability",
      "title": "Assess methods for protein stability experiments",
      "question": "What evidence supports ranking protein substitutions by folding stability, and what must be validated before choosing a method?",
      "area": "proteins-complexes",
      "contexts": [
        "research",
        "clinical_research"
      ],
      "search_terms": [
        "protein variant",
        "amino acid substitution",
        "folding stability",
        "protein engineering",
        "proteolysis",
        "AMFR",
        "ProteinGym",
        "ESM-2",
        "EVmutation",
        "EVCouplings",
        "pathogenicity"
      ],
      "intended_users": [
        "Experimental researchers selecting variants for protein stability experiments",
        "Computational researchers evaluating sequence-based variant rankings",
        "Translational researchers checking whether stability evidence answers a disease question"
      ],
      "decision": "Inspect the available AMFR assay results as a narrow example, then identify the missing singles-only comparison and validation required for the target protein and endpoint.",
      "inputs": [
        "Wild-type protein sequence and amino acid substitutions",
        "A defined construct and experimental stability endpoint; alignment-derived methods additionally require a traceable alignment and model"
      ],
      "output": "Exact existing AMFR configurations, their assay scope and unresolved comparisons; no universal model ranking or pathogenicity prediction.",
      "setting": "The current evidence is limited to the 47-residue AMFR_HUMAN_Tsuboyama_2023_4G3O construct in ProteinGym v1.3. Its cDNA-display proteolysis assay infers folding stability. Completed evaluations cover 2,972 variants: 820 single and 2,152 double substitutions.",
      "exclusions": [
        "Whole-protein function, cellular activity, organismal fitness and clinical pathogenicity",
        "Generalisation to other proteins, other ESM-2 checkpoints or the full ProteinGym track",
        "Treating the existing mixed cohort as a completed matched single-substitution comparison"
      ],
      "clinical_scope": "Clinical applicability is not established. Stability of this short experimental construct is not evidence of clinical pathogenicity or suitability for diagnosis or treatment.",
      "evidence_gaps": [
        "The existing ESM-2 and fixed-seed random results use separate protocols. They are displayed separately and do not establish a matched cross-protocol winner.",
        "No uncertainty intervals or seed-variability estimates are recorded for these completed evaluations. One random ranking is not a chance-performance interval.",
        "The proposed ESM-2 versus EVCouplings site-independent and EVmutation comparison targets a frozen set of covered single substitutions. It has no completed results.",
        "The planned comparison still needs a separate execution decision, a traceable real evolutionary model, the full-model adapter, resource logging, frozen populations and tested analysis code. Planning resource ceilings are not measured requirements.",
        "The recorded ESM-2 timer excludes checkpoint loading, preparation and metrics; peak memory is unreported. It is not an end-to-end or cross-model speed comparison.",
        "The existing AMFR protocols have no reviewed direct task-membership link. The mappings therefore reference the protocols only and do not infer membership from the ProteinGym suite.",
        "Assay bytes were hashed locally without independent authentication against an upstream published checksum. Training overlap, independent reproduction and human scientific review remain unresolved."
      ],
      "citations": [
        {
          "source_id": "profile-protocol-proteingym-reference-files-dms-substitutions-csv-a8f49801",
          "locator": "DMS_id=AMFR_HUMAN_Tsuboyama_2023_4G3O: seq_len, selection_assay, selection_type, raw_DMS_phenotype_name, DMS_number_single_mutants, DMS_number_multiple_mutants"
        },
        {
          "source_id": "rewire-local-20260920-source-proteingym-esm2",
          "locator": "/coverage; /execution; /provenance; /protocol_results"
        },
        {
          "source_id": "rewire-local-20260921-source-proteingym-random",
          "locator": "/coverage; /model_configuration; /protocol_results"
        },
        {
          "source_id": "use-case-source-amfr-pilot-plan-194a78b",
          "locator": "Status; Question; Task and data; Methods; Gates and execution order; Runtime and cost; Roles and review"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-25T15:43:22Z",
        "note": "Bounded review of pinned assay metadata, completed execution records and the planning artifact. The plan is not a result. No new model execution, human domain review, independent replication or clinical validation. Public planned-work navigation points to the exact reviewed source copy; the original private-repository link remains in source provenance."
      },
      "planned_work": [
        {
          "title": "Matched single-substitution comparison: ESM-2, site-independent model and EVmutation",
          "url": "https://benchmarks.rewire.it/omics/sources/a3208aa4537e8d28c56f68f4559299f31241c50b8d8b2b3776cba5f1718caa43.md",
          "status": "blocked",
          "reason": "Planning is complete; execution is not authorised by that plan. Gates require an execution decision, traceable evolutionary model, full-model adapter, resource logging, frozen populations and tested analysis. No measured comparison or winner is available."
        }
      ]
    },
    {
      "id": "use-case-rare-disease-candidate-ranking",
      "slug": "rare-disease-candidate-ranking",
      "title": "Rank rare-disease variants for review",
      "question": "Which methods recover causal variants and genes within a realistic laboratory review budget?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "rare disease",
        "variant prioritisation",
        "gene ranking",
        "exome sequencing",
        "genome sequencing",
        "phenotype",
        "pedigree",
        "Exomiser"
      ],
      "intended_users": [
        "Diagnostic scientists",
        "Clinical genomicists evaluating interpretation workflows"
      ],
      "decision": "Choose methods for prioritising variants in an initial exome or genome analysis, using phenotype, pedigree and molecular evidence to focus laboratory review.",
      "inputs": [
        "Quality-controlled variant calls, genome build and calling coverage",
        "Phenotype terms, pedigree, inheritance and available family sequencing",
        "Dated population-frequency and gene–disease evidence"
      ],
      "output": "A ranked candidate list with supporting evidence, conflicts, filter reasons and unresolved findings for professional review.",
      "setting": "Initial ES/GS interpretation in a defined congenital-anomaly or developmental-disorder population. Singleton and family-based analysis require separate comparisons.",
      "exclusions": [
        "Variant ranking alone does not establish pathogenicity or a patient diagnosis.",
        "Balanced pathogenic/benign variant classification does not measure case-level diagnostic performance.",
        "Reanalysis, cancer-risk estimation and treatment selection are separate questions."
      ],
      "clinical_scope": "Clinical research on interpretation support. Confirmed diagnoses and downstream management outcomes require separate evaluation from candidate retrieval.",
      "evidence_gaps": [
        "Case-level comparisons using identical inputs and a fixed review budget remain to be collected.",
        "Independent qualified clinical-genetics adjudication, unresolved cases and variant-calling failures are needed to assess clinical performance.",
        "Published validation summaries do not establish access to reusable patient-level cohorts."
      ],
      "citations": [
        {
          "source_id": "use-case-source-clinical-priorities-2026-09-28",
          "locator": "C1 — Rare-disease candidate ranking"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex",
        "reviewed_at": "2026-09-28T14:52:02.000Z",
        "note": "Automated review of Rewire's workflow definition and sourced research brief against the agreed exploration and backlog. No human expert approval, model-applicability assessment or clinical validation is claimed."
      },
      "collection_plan": {
        "status": "planned",
        "comparison_question": "Does adding a molecular-effect model to phenotype-, pedigree- and frequency-aware analysis improve causal-finding recovery at the same analyst review budget?",
        "baselines": [
          "Conventional filtering and phenotype/pedigree-aware review with pinned versions",
          "An appropriate pinned Exomiser configuration without the added model"
        ],
        "outcomes": [
          "Case-level recovery of independently adjudicated causal findings within the review budget",
          "Analyst effort, missed variant classes, unresolved cases and abstention",
          "Confirmed diagnoses measured separately from ranked candidates"
        ],
        "validation_requirements": [
          "Separate families, centres and evaluation time periods.",
          "Freeze knowledge releases and model training cutoffs; audit predictor-derived labels.",
          "Retain unresolved cases in workload and coverage denominators.",
          "Record calling failures separately and include them in end-to-end sensitivity.",
          "Use independent qualified clinical-genetics adjudication for diagnostic claims."
        ],
        "next_step": "Build a case-level evidence table separating solved-case ranking, unselected diagnostic evaluation and molecular-effect proxies, with exact versions and same-input comparators."
      },
      "planned_work": []
    },
    {
      "id": "use-case-regulatory-variant-gene-follow-up",
      "slug": "regulatory-variant-gene-follow-up",
      "title": "Select regulatory variants and genes for functional follow-up",
      "question": "Which variants, regulatory elements and genes should I perturb to explain a disease-associated locus?",
      "area": "dna-genomes",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "regulatory variant",
        "effector gene",
        "variant-to-gene",
        "enhancer-to-gene",
        "GWAS",
        "fine mapping",
        "CRISPRi",
        "MPRA",
        "endogenous editing",
        "AlphaGenome",
        "ENCODE",
        "scE2G"
      ],
      "intended_users": [
        "Functional geneticists",
        "Researchers interpreting disease-associated loci"
      ],
      "decision": "Choose allele edits, regulatory-element perturbations and gene readouts that can distinguish competing explanations for a locus.",
      "inputs": [
        "Fine-mapped variants with genome build, ancestry and linkage-disequilibrium context",
        "Candidate regulatory elements, genes and a relevant cell type",
        "Sequence and available chromatin, expression or contact measurements",
        "The feasible perturbation assay, readout and follow-up budget"
      ],
      "output": "A prioritised set of variant–element–gene hypotheses, with the experiments needed to test each link and effect direction where supported.",
      "setting": "Research follow-up of defined disease-associated loci in a specified cell system. Allele-effect prediction, element–gene linking and experimental shortlist selection are separate comparisons.",
      "exclusions": [
        "Reporter activity does not by itself establish endogenous regulation.",
        "Perturbing an entire regulatory element is different from editing one allele.",
        "A regulatory link alone does not establish a gene’s causal role in disease."
      ],
      "clinical_scope": "Research only. Molecular effects and regulatory links do not establish clinical variant classification or diagnostic validity.",
      "evidence_gaps": [
        "Endpoint-specific comparisons must be collected for allele effects, element–gene links and experimental selection.",
        "Evidence from the intended cell context and endogenous perturbation may be missing even when reporter or linking benchmarks exist.",
        "Human domain review remains unassigned."
      ],
      "citations": [
        {
          "source_id": "use-case-source-research-priorities-2026-09-28",
          "locator": "R2 — Regulatory variant and effector-gene follow-up"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex",
        "reviewed_at": "2026-09-28T14:52:13Z",
        "note": "Automated review of the workflow definition, cited primary-source scope and collection plan. No model evaluation or applicability mapping was added. Human domain review remains unassigned."
      },
      "collection_plan": {
        "status": "planned",
        "comparison_question": "Does a method select more experimentally supported regulatory hypotheses than compatible distance or linking baselines at the same follow-up budget?",
        "baselines": [
          "Distance-based variant/gene or enhancer/gene selection where applicable",
          "ABC, ENCODE-rE2G or scE2G for compatible element–gene linking tasks",
          "Sequence predictors compared only on the allele-effect outputs they produce"
        ],
        "outcomes": [
          "Allele-dependent regulatory effects",
          "Endogenous element–gene and allele–gene effects, measured separately",
          "Confirmed follow-up hits per assay budget"
        ],
        "validation_requirements": [
          "Define the allele-effect, linking or selection endpoint before choosing a comparison.",
          "Hold out loci and studies; audit linkage-disequilibrium and pretraining overlap.",
          "Preserve cell context, genome build, assay and perturbation type.",
          "Keep reporter activity, whole-element perturbation, single-base effects and disease causality distinct."
        ],
        "next_step": "Build separate extraction tables for CRISPRi linking, allele-specific reporter and endogenous-editing studies, identifying which test experimental selection."
      },
      "planned_work": []
    },
    {
      "id": "use-case-rhodopsin-wavelength-transfer",
      "slug": "rhodopsin-wavelength-transfer",
      "title": "Assess rhodopsin wavelength prediction across sequence backgrounds",
      "question": "Which sequence-based configurations merit testing before selecting rhodopsins with different absorption wavelengths in a new wild-type background?",
      "area": "proteins-complexes",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "rhodopsin",
        "opsin",
        "spectral tuning",
        "absorption wavelength",
        "optogenetics",
        "protein engineering",
        "FLIP2",
        "Rhomax",
        "wild-type transfer",
        "ESM-2",
        "sequence composition",
        "ridge regression"
      ],
      "intended_users": [
        "Protein engineers planning rhodopsin spectral-tuning experiments",
        "Computational researchers checking transfer to new sequence backgrounds"
      ],
      "decision": "Use the existing held-out-background evaluations to choose controls and exact model configurations for a new, prospectively held-out spectral-tuning experiment.",
      "inputs": [
        "Complete rhodopsin amino-acid sequences",
        "Training measurements of peak absorption wavelength and frozen assignments that separate wild-type backgrounds"
      ],
      "output": "Evidence for the exact frozen sequence probes and controls, plus the validation still needed for a new background; no calibrated wavelength or wet-lab success guarantee.",
      "setting": "One complete FLIP2 Rhomax by_wild_type test split: 584 training, 116 unused validation and 184 test records. The experimental endpoint is peak absorption wavelength in nanometres.",
      "exclusions": [
        "Rhodopsin activation efficiency, expression, photostability and cellular function",
        "Zero-shot likelihood scoring, alternative pooling or fine-tuning not evaluated in these runs",
        "General protein fitness, all FLIP2 landscapes and clinical use"
      ],
      "clinical_scope": "Research only. Absorption-wavelength ranking does not establish suitability for an optogenetic intervention, patient care or treatment.",
      "evidence_gaps": [
        "A spectral-tuning endpoint is not opsin activation efficiency, expression, photostability, cellular function, general protein fitness or clinical usefulness.",
        "Spearman and full-ranking NDCG do not establish wavelength calibration, top-k precision or a prospective experimental hit rate. Constant predictions have undefined Spearman; their NDCG is a control value, not strong predictive evidence.",
        "No intervals or seed-variability estimates. The 35M checkpoint was chosen after seeing the 8M outcome; the source describes an exploratory follow-up, not a preregistered family-scale comparison.",
        "ESM-2 pretraining overlap is unresolved. A frozen linear probe result does not establish the value of alternative pooling, fine-tuning or every model-family configuration.",
        "Recorded timing covers sections of execution and is not a hardware-normalised deployment-cost comparison. Human domain review and external replication are not established."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260921-instructions-esm2-35m",
          "locator": "Matched Rhomax evaluations; Procedure and evidence; Re-run or inspect"
        },
        {
          "source_id": "rewire-local-20260921-source-esm2-35m",
          "locator": "/metrics; /coverage; /model_configuration; /protocol_configuration; /execution; /provenance"
        },
        {
          "source_id": "rewire-local-20260921-source-esm2-8m",
          "locator": "/metrics; /coverage; /model_configuration; /protocol_configuration; /execution; /provenance"
        },
        {
          "source_id": "rewire-local-20260921-source-composition22",
          "locator": "/metrics; /model_configuration; /protocol_configuration"
        },
        {
          "source_id": "rewire-local-20260920-source-flip2-train-mean",
          "locator": "/metrics; /coverage; /model_configuration"
        },
        {
          "source_id": "rewire-local-20260920-source-flip2-composition",
          "locator": "/metrics; /coverage; /model_configuration; /provenance"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "planned_work": []
    },
    {
      "id": "use-case-somatic-small-variant-oncogenicity",
      "slug": "somatic-small-variant-oncogenicity",
      "title": "Assess somatic small-variant oncogenicity",
      "question": "Which methods help classify somatic SNVs and small indels while preserving uncertainty and the relevant gene mechanism?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "somatic variant",
        "oncogenicity",
        "SNV",
        "small indel",
        "oncogene",
        "tumour suppressor",
        "ClinGen",
        "CGC",
        "VICC"
      ],
      "intended_users": [
        "Somatic variant curators",
        "Molecular pathologists evaluating interpretation methods"
      ],
      "decision": "Choose methods that support evidence-based oncogenicity review of somatic small variants and identify when the available evidence remains insufficient.",
      "inputs": [
        "Variant, genome build, transcript, allele fraction, quality and confidence in somatic origin",
        "Tumour context, gene mechanism and dated oncogenicity-specific assertions",
        "Functional studies and the applicable criteria specification"
      ],
      "output": "An oncogenicity classification or unresolved assessment with constituent criteria, conflicting findings and evidence needs.",
      "setting": "SNVs and small indels, with separate protocols for oncogene missense and tumour-suppressor loss-of-function mechanisms.",
      "exclusions": [
        "Inherited predisposition, fusions, rearrangements and copy-number variants require separate interpretation protocols.",
        "Oncogenicity and molecular activity do not establish drug response or clinical actionability.",
        "Germline-benign labels are not automatically valid somatic negative controls."
      ],
      "clinical_scope": "Clinical research on somatic interpretation support. Tumour-only sequencing does not establish somatic origin, and therapeutic relevance needs a separate context-specific review.",
      "evidence_gaps": [
        "Independent oncogenicity assertions and orthogonal functional evidence remain to be assembled with uncertain and negative examples.",
        "Qualified somatic reviewers are needed to adjudicate criteria and evidence separately from the method being tested.",
        "Older functional datasets and curated assertions may overlap model training or include predictor-derived evidence."
      ],
      "citations": [
        {
          "source_id": "use-case-source-clinical-priorities-2026-09-28",
          "locator": "C4 — Somatic small-variant oncogenicity"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex",
        "reviewed_at": "2026-09-28T14:52:02.000Z",
        "note": "Automated review of Rewire's workflow definition and sourced research brief against the agreed exploration and backlog. No human expert approval, model-applicability assessment or clinical validation is claimed."
      },
      "collection_plan": {
        "status": "planned",
        "comparison_question": "Does a proposed method improve independent oncogenicity classification or review efficiency over explicit conventional criteria while retaining calibrated unresolved results?",
        "baselines": [
          "ClinGen/CGC/VICC criteria-based review under a pinned specification",
          "A relevant specialised predictor assessed only within its actual output scope"
        ],
        "outcomes": [
          "Category precision/recall and criteria fidelity",
          "Calibration, abstention and unresolved-case coverage",
          "Expert review effort, measured separately from treatment outcomes"
        ],
        "validation_requirements": [
          "Use oncogenicity-specific labels with evidence, review status and dates.",
          "Hold out allelic series, residues, studies and genes where unseen-gene transfer is intended.",
          "Audit training and assertion overlap, including predictor contributions to labels.",
          "Preserve tumour, assay and gene-mechanism context.",
          "Use independent qualified somatic-variant adjudication and justified negative or uncertain examples."
        ],
        "next_step": "Assemble a provenance table of oncogenicity assertions and orthogonal functional studies, including label-leakage exclusions and independently reviewable uncertain examples."
      },
      "planned_work": []
    },
    {
      "id": "use-case-splicing-follow-up",
      "slug": "splicing-follow-up",
      "title": "Prioritise variants for splicing experiments",
      "question": "Which evaluated configurations can inform selection of human SNVs for follow-up splicing experiments?",
      "area": "dna-genomes",
      "contexts": [
        "research",
        "clinical_research"
      ],
      "search_terms": [
        "splicing",
        "splice disruption",
        "SNV",
        "single nucleotide variant",
        "variant prioritisation",
        "exon recognition",
        "minigene",
        "patient RNA",
        "MFASS",
        "SpliceAI",
        "Pangolin"
      ],
      "intended_users": [
        "Experimental researchers selecting variants for functional follow-up",
        "Computational researchers comparing splice-effect configurations",
        "Clinical researchers investigating the limits of assay evidence"
      ],
      "decision": "Inspect the matched MFASS configurations and their limitations before selecting a method and designing validation in the intended experimental setting.",
      "inputs": [
        "Human single nucleotide variants with alleles, genome assembly and gene/transcript context",
        "The intended experimental endpoint and the number of variants that can be followed up"
      ],
      "output": "A sourced set of exact evaluated configurations and remaining validation needs; no patient-level variant classification or recommendation.",
      "setting": "MFASS measures exon recognition in an artificial minigene reporter. The matched study evaluates genomic-context SpliceAI and Pangolin configurations against this functional endpoint on a fixed held-out population.",
      "exclusions": [
        "Indels and variant types outside the reviewed human SNV protocol",
        "Patient-RNA effects, disease pathogenicity, diagnostic yield and treatment decisions",
        "Pooling the matched annotation study with historical MFASS runs using other annotations or scoring populations"
      ],
      "clinical_scope": "Clinical applicability is not established. Reporter-assay ranking does not demonstrate patient-RNA performance, pathogenicity classification or clinical yield; validation in the intended population and workflow is still needed.",
      "evidence_gaps": [
        "All four conditions scored 8,297 of 8,324 held-out variants. The same 27 exclusions comprise 23 hg19-to-hg38 assembly-orientation mismatches and four canonical-transcript-span exclusions; the latter are not established faulty variants. Missing predictions are not negative predictions.",
        "No top-100 precision difference is established; Pangolin's masked top-100 result is sensitive to the registered tie order. Precision at 100 does not transfer automatically to another follow-up capacity or prevalence.",
        "Individual-condition uncertainty intervals are not recorded. Paired-contrast intervals concern differences between conditions and must not be shown as each condition's uncertainty.",
        "The study is exploratory: prior outcomes were inspected and nine contrast intervals are unadjusted. Matching annotation does not isolate model architecture.",
        "Author confirmation of the assembly-orientation finding is not established. Human scientific review and independent replication remain outstanding.",
        "These mappings do not change the MFASS task's discovered status. A source-reviewed applicability mapping is separate from reviewing the task record."
      ],
      "citations": [
        {
          "source_id": "evidence-expansion-mfass-readme-62a93814",
          "locator": "Dataset; Limits: artificial minigene exon-recognition endpoint, not patient RNA"
        },
        {
          "source_id": "use-case-source-mfass-matched-intake-194a78b",
          "locator": "Opening paragraphs: coverage, exclusion classes, annotation, review and interpretation; Review and validation"
        },
        {
          "source_id": "rewire-mfass-matched-v1-source-report",
          "locator": "/conditions/{S0,S1,P0,P1}/coverage; /conditions/{S0,S1,P0,P1}/ties; /contrasts/{S1-S0,P1-P0,P0-S0}/paired/precision_at_capacity"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-25T15:28:11Z",
        "note": "Bounded review of pinned reports, protocol documentation and intake narrative. No new model execution, human domain review, independent replication or clinical validation."
      },
      "planned_work": []
    },
    {
      "id": "use-case-structural-hypotheses-experiments",
      "slug": "structural-hypotheses-experiments",
      "title": "Choose structural hypotheses to guide experiments",
      "question": "Which predicted interfaces or structures are reliable enough to guide my next experiment?",
      "area": "proteins-complexes",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "structural hypothesis",
        "protein complex",
        "interface mutation",
        "construct selection",
        "protein structure",
        "experimental planning",
        "AlphaFold",
        "FoldBench",
        "CASP",
        "CAPRI",
        "confidence calibration"
      ],
      "intended_users": [
        "Structural biologists",
        "Researchers selecting interface mutations or protein constructs"
      ],
      "decision": "Choose structural hypotheses and associated mutations or constructs for experimental testing.",
      "inputs": [
        "Protein sequences, partner identities and the intended assembly",
        "Available experimental structures or templates with release dates",
        "A defined experimental choice and testing budget",
        "Relevant biochemical conditions and alternative conformations"
      ],
      "output": "Structural hypotheses with confidence and coverage estimates, linked to proposed mutations or constructs and tests of their reliability.",
      "setting": "Research planning within a bounded class of protein complexes. Monomers, antibody complexes and ligand complexes require separate comparison protocols.",
      "exclusions": [
        "Structural accuracy alone does not establish that a proposed experiment will succeed.",
        "High confidence does not prove interaction existence, affinity or function.",
        "Results from one complex class do not establish performance in another."
      ],
      "clinical_scope": "Research only. Structural hypotheses do not establish therapeutic efficacy or clinical suitability.",
      "evidence_gaps": [
        "Task-specific structural comparisons and evidence of experimental selection benefit must be collected separately.",
        "Independent mutation or construct outcomes may be unavailable even when structural benchmarks exist.",
        "Human domain review remains unassigned."
      ],
      "citations": [
        {
          "source_id": "use-case-source-research-priorities-2026-09-28",
          "locator": "R5 — Structural hypotheses for experiments"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex",
        "reviewed_at": "2026-09-28T14:52:13Z",
        "note": "Automated review of the workflow definition, cited primary-source scope and collection plan. No model evaluation or applicability mapping was added. Human domain review remains unassigned."
      },
      "collection_plan": {
        "status": "planned",
        "comparison_question": "Do model-derived structural hypotheses improve experimental choices over appropriate conventional strategies at the same testing budget?",
        "baselines": [
          "Suitable template modelling or docking for the specified complex class",
          "Conventional mutation or construct selection for experimental-utility comparisons"
        ],
        "outcomes": [
          "Task-specific contact or pose accuracy and prediction failures as intermediate evidence",
          "Confidence calibration and coverage",
          "Independent interface-mutation or construct success where measured"
        ],
        "validation_requirements": [
          "Define the experimental choice and compatible complex class before selecting metrics.",
          "Audit sequence, template, ligand and structure-deposition overlap with training data.",
          "Include failed predictions and relevant alternative conformations.",
          "Keep structural accuracy, interaction existence, affinity, function and experimental utility distinct."
        ],
        "next_step": "Build a task-specific structural benchmark table and a separate inventory of independent mutation or construct outcomes."
      },
      "planned_work": []
    },
    {
      "id": "use-case-therapeutic-target-validation",
      "slug": "therapeutic-target-validation",
      "title": "Select therapeutic targets for validation",
      "question": "Which targets should I test to change a defined disease-relevant phenotype?",
      "area": "cells-tissues",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "target prioritisation",
        "target validation",
        "therapeutic target",
        "disease mechanism",
        "direction of effect",
        "genetic dependency",
        "Open Targets",
        "DepMap"
      ],
      "intended_users": [
        "Disease biologists",
        "Translational discovery teams"
      ],
      "decision": "Select targets and modulation directions for a validation batch, with clear tests of the proposed disease mechanism.",
      "inputs": [
        "A defined disease, biological context and experimental phenotype",
        "A candidate gene set and feasible ways to inhibit or activate each target",
        "Dated genetic, expression and functional evidence, including conflicting findings",
        "A validation budget and relevant normal-cell or selectivity controls"
      ],
      "output": "A shortlist of target–disease–intervention hypotheses with supporting evidence, uncertainties and proposed validation experiments.",
      "setting": "Research planning for one disease and a prespecified experimental system. The intended comparison tests whether rankings improve the yield of useful target effects.",
      "exclusions": [
        "Untested targets cannot be treated as negative examples.",
        "Cellular dependency, target tractability and a therapeutic window require different evidence.",
        "A target association does not establish that modulating it will benefit patients."
      ],
      "clinical_scope": "Research only. Target prioritisation does not establish treatment benefit or support patient-specific treatment decisions.",
      "evidence_gaps": [
        "Comparisons need independent validation campaigns that report failures as well as successful targets.",
        "Matched-budget evidence for phenotype effects, rescue and selectivity has not yet been collected for this question.",
        "Human domain review remains unassigned."
      ],
      "citations": [
        {
          "source_id": "use-case-source-research-priorities-2026-09-28",
          "locator": "R1 — Therapeutic target validation"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex",
        "reviewed_at": "2026-09-28T14:52:13Z",
        "note": "Automated review of the workflow definition, cited primary-source scope and collection plan. No model evaluation or applicability mapping was added. Human domain review remains unassigned."
      },
      "collection_plan": {
        "status": "planned",
        "comparison_question": "Does a proposed ranking identify more reproducible disease-relevant target effects than conventional evidence aggregation at the same validation budget?",
        "baselines": [
          "Genetics-only and expression-only rankings",
          "Simple dependency ranking or conventional evidence aggregation"
        ],
        "outcomes": [
          "Confirmed useful target effects among all candidates tested",
          "Reproducibility, rescue and context/selectivity",
          "Failed experiments and uncertainty about modulation direction"
        ],
        "validation_requirements": [
          "Define the disease, candidate universe, intervention direction, phenotype and budget before comparing rankings.",
          "Freeze input evidence before validation outcomes and audit overlap between discovery and validation data.",
          "Include failed experiments and normal-cell controls; retain untested candidates as untested.",
          "Keep target association, dependency, tractability and clinical success separate."
        ],
        "next_step": "Build a disease-specific inventory of validation campaigns, including failed targets, evidence dates and conventional ranking baselines."
      },
      "planned_work": []
    },
    {
      "id": "use-case-unresolved-rare-disease-reanalysis",
      "slug": "unresolved-rare-disease-reanalysis",
      "title": "Reanalyse unresolved rare-disease cases",
      "question": "Which reanalysis methods find new diagnoses or reduce review effort when given the same updated information?",
      "area": "dna-genomes",
      "contexts": [
        "clinical_research"
      ],
      "search_terms": [
        "rare disease",
        "reanalysis",
        "negative exome",
        "unresolved genome",
        "variant reevaluation",
        "Talos",
        "Exomiser",
        "ClinVar"
      ],
      "intended_users": [
        "Clinical genomics services responsible for unresolved cases",
        "Laboratory reanalysis leads"
      ],
      "decision": "Choose a workflow that identifies previously unresolved cases worth reopening and explains the evidence changes that warrant review.",
      "inputs": [
        "Original ES/GS calls, coverage, report, filters and analysis date",
        "Updated phenotype, pedigree and gene–disease or variant evidence",
        "Dated changes to calling and interpretation methods"
      ],
      "output": "A prioritised queue of new or changed findings, their evidence history and the review or confirmation needed to resolve each case.",
      "setting": "A defined cohort remaining unresolved after an earlier ES/GS analysis, followed over a stated interval. Variant reevaluation and whole-case reanalysis are recorded separately.",
      "exclusions": [
        "Reranking known solved cases does not establish new diagnostic yield.",
        "New knowledge, phenotypes or variant calls must not be counted automatically as algorithmic improvement.",
        "A changed database label or new candidate is not a confirmed diagnosis."
      ],
      "clinical_scope": "Clinical research on reanalysis support. Programme-level yield and the added value of a method require different comparisons; diagnoses and diagnosed individuals have separate denominators.",
      "evidence_gaps": [
        "Dated programme outcomes and comparisons against refreshed conventional analysis remain to be collected.",
        "Independent qualified review must confirm new findings and investigate false alerts or retracted diagnoses.",
        "Published implementation evidence does not establish a universal reanalysis interval or access to controlled cohorts."
      ],
      "citations": [
        {
          "source_id": "use-case-source-clinical-priorities-2026-09-28",
          "locator": "C2 — Unresolved rare-disease reanalysis"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex",
        "reviewed_at": "2026-09-28T14:52:02.000Z",
        "note": "Automated review of Rewire's workflow definition and sourced research brief against the agreed exploration and backlog. No human expert approval, model-applicability assessment or clinical validation is claimed."
      },
      "collection_plan": {
        "status": "planned",
        "comparison_question": "Does a proposed reanalysis method improve confirmed diagnoses or reduce review burden relative to refreshed conventional analysis using identical updated knowledge, phenotypes and calls?",
        "baselines": [
          "Refreshed conventional analysis with the same updated information",
          "The original unresolved cohort for programme-level incremental yield, reported separately"
        ],
        "outcomes": [
          "New confirmed diagnoses per eligible case over a stated interval",
          "Added method benefit under equal updated inputs",
          "False alerts, review time and the reason for each resolved case"
        ],
        "validation_requirements": [
          "Record original and update dates; exclude future knowledge at each analysis time.",
          "Separate benefit from new information, calling changes and prioritisation changes.",
          "Audit family/site overlap and originating-laboratory label circularity.",
          "Count diagnoses and diagnosed individuals separately; distinguish first and repeated cycles.",
          "Use independent qualified clinical-genetics review to confirm outcomes."
        ],
        "next_step": "Create a dated reanalysis evidence table with separate programme-yield and method-comparison columns, noting where an equal-input comparator is missing."
      },
      "planned_work": []
    },
    {
      "id": "use-case-utr-translation-baselines",
      "slug": "utr-translation-baselines",
      "title": "Set baselines for UTR translation experiments",
      "question": "Before training a more complex model of reporter translation, which sequence-only controls should be measured on the same held-out data?",
      "area": "rna-transcriptomes",
      "contexts": [
        "research"
      ],
      "search_terms": [
        "5 prime UTR",
        "5′UTR",
        "translation",
        "mean ribosome load",
        "MRL",
        "reporter assay",
        "synthetic mRNA",
        "mRNABench",
        "Sample",
        "RidgeCV",
        "sequence composition",
        "training mean",
        "baseline"
      ],
      "intended_users": [
        "Computational researchers building models of reporter translation",
        "Experimental researchers evaluating a designed 5′UTR library before selecting a modelling strategy"
      ],
      "decision": "Establish training-mean and sequence-composition controls on a fixed held-out reporter dataset before deciding whether a more complex model adds useful prediction. The current evidence contains these two procedural controls and no pretrained model.",
      "inputs": [
        "Complete processed sequences and measured mean ribosome load from a defined reporter library",
        "A fixed training, validation and test assignment, with test labels withheld from fitting and model selection"
      ],
      "output": "Two exact baseline configurations, their matched held-out evaluations and a repeat recipe; no foundation-model ranking or prediction of therapeutic performance.",
      "setting": "The current comparison covers only the Sample designed subset in mRNABench and its measured mean ribosome load. Both controls use the full processed source sequence and the same seed-2541 split: 70,011 training, 15,003 validation and 15,003 test records. Both score all 15,003 held-out test records.",
      "exclusions": [
        "Benchmark-wide mRNABench performance or comparisons with the paper’s aggregated model scores",
        "Claims about novel combinations of sequence motifs, homology-separated generalisation, other reporter systems or RNA chemistries",
        "Therapeutic potency, in-vivo protein output and clinical decisions"
      ],
      "clinical_scope": "This is research evidence for designing a baseline comparison. Reporter mean ribosome load does not establish therapeutic efficacy or clinical suitability.",
      "evidence_gaps": [
        "Only two procedural controls have been evaluated here; no pretrained model or more complex sequence model has a matched result in this protocol.",
        "There is one split and one execution per configuration, with no uncertainty interval or seed-variability estimate. The split does not establish homology separation or compositional generalisation.",
        "RidgeCV uses training labels only and leaves the validation split unused. This differs from upstream probing; the selected test MSE must not be compared as if it were the paper’s aggregated Pearson score or default validation result.",
        "The training-mean control has undefined Pearson and Spearman correlations because its predictions are constant. Unavailable correlations are not zero performance scores.",
        "The original run’s raw inputs and saved predictions are not publicly archived. The public reports record hashes and a recipe for obtaining source data and running the controls again; prior predictions cannot be rescored from those reports. Dataset reuse terms are unreported.",
        "The recorded repeat recipe executes all four local sequence controls, including a separate protein dataset. Portable command examples have not been rerun verbatim during this review. Timings cover the reported calculation sections, not complete setup or cross-machine performance."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260920-source-mrnabench-composition",
          "locator": "/coverage; /model_configuration; /protocol_configuration; /protocol_results; /provenance"
        },
        {
          "source_id": "rewire-local-20260920-source-mrnabench-train-mean",
          "locator": "/coverage; /model_configuration; /protocol_results/metric_unavailable_reasons; /provenance"
        },
        {
          "source_id": "rewire-local-20260920-instructions-mrnabench",
          "locator": "Results: RNA evaluation paragraph; Provenance and review; Reproduce the four sequence controls"
        },
        {
          "source_id": "expansion-p3-mrnabench-2025",
          "locator": "Local Tasks (S12); Appendix B.7 Mean Ribosome Load - MPRA (S33); Data Splitting Strategies (S17); Linear Probing (S18); Appendix C (APP3)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "planned_work": []
    }
  ],
  "mappings": [
    {
      "id": "use-case-map-gears-norman-table6-mse",
      "use_case_id": "use-case-genetic-perturbation-response",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "gears-2023-supp-table6-task-mse",
      "evaluation_ids": [
        "gears-2023-supp-table6-evaluation-no-perturb-mse",
        "gears-2023-supp-table6-evaluation-cpa-mse",
        "gears-2023-supp-table6-evaluation-cpa-plus-kg-mse",
        "gears-2023-supp-table6-evaluation-gears-mse"
      ],
      "endpoint": "Mean squared error of predicted versus observed post-perturbation expression; lower is better.",
      "relevance": "proxy",
      "rationale": "The complete comparison includes a no-change control, CPA, CPA with knowledge-graph features and GEARS. It can inform which controls to include in a pilot; aggregate expression agreement does not directly measure follow-up experiment yield.",
      "constraints": [
        "Use the exact Norman2019 K562 source-table protocol and configurations; do not combine this evidence with PerturBench or other GEARS splits.",
        "Read both Table 6 endpoints together while retaining their separate metrics and shared experimental provenance."
      ],
      "limitations": [
        "Exact Table 6 split and scored gene subset remain unextracted; checkpoint and runtime requirements are not established by these catalogue evaluations.",
        "Reported spreads have unspecified type. A numerical difference is not a significance test.",
        "Predictions need validation in the intended cell system before experiment selection."
      ],
      "citations": [
        {
          "source_id": "coverage-source-gears-supp",
          "locator": "Supplementary Table 6, printed page 34 / PDF page 35; Supplementary Table 1, PDF page 30; Supplementary Notes 5 and 14 (K562 context)."
        },
        {
          "source_id": "evidence-official-c037e3419c04936262a0",
          "locator": "README.md at f374e43e197b295016d80395d7a54ddb81cc6769, A note on usage (lines 15–19) and Core API Interface (lines 20–54)."
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "c28653a778ad2953ce30515281cfbcbec052ef644991079b2d503d7ae92c22c2"
    },
    {
      "id": "use-case-map-gears-norman-table6-pearson-de",
      "use_case_id": "use-case-genetic-perturbation-response",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "gears-2023-supp-table6-task-pearson-de",
      "evaluation_ids": [
        "gears-2023-supp-table6-evaluation-no-perturb-pearson-de",
        "gears-2023-supp-table6-evaluation-cpa-pearson-de",
        "gears-2023-supp-table6-evaluation-cpa-plus-kg-pearson-de",
        "gears-2023-supp-table6-evaluation-gears-pearson-de"
      ],
      "endpoint": "Pearson correlation of predicted versus observed expression change from unperturbed controls; higher is better.",
      "relevance": "proxy",
      "rationale": "The complete comparison includes a no-change control, CPA, CPA with knowledge-graph features and GEARS. It can inform which controls to include in a pilot; aggregate expression agreement does not directly measure follow-up experiment yield.",
      "constraints": [
        "Use the exact Norman2019 K562 source-table protocol and configurations; do not combine this evidence with PerturBench or other GEARS splits.",
        "Read both Table 6 endpoints together while retaining their separate metrics and shared experimental provenance."
      ],
      "limitations": [
        "Exact Table 6 split and scored gene subset remain unextracted; checkpoint and runtime requirements are not established by these catalogue evaluations.",
        "Reported spreads have unspecified type. A numerical difference is not a significance test.",
        "Predictions need validation in the intended cell system before experiment selection."
      ],
      "citations": [
        {
          "source_id": "coverage-source-gears-supp",
          "locator": "Supplementary Table 6, printed page 34 / PDF page 35; Supplementary Table 1, PDF page 30; Supplementary Notes 5 and 14 (K562 context)."
        },
        {
          "source_id": "evidence-official-c037e3419c04936262a0",
          "locator": "README.md at f374e43e197b295016d80395d7a54ddb81cc6769, A note on usage (lines 15–19) and Core API Interface (lines 20–54)."
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "87737c4787ced71240c112f2f4cc859131132633b9f029abb0d8aa043f310486"
    },
    {
      "id": "use-case-mapping-msms-formula-no-formula-r1",
      "use_case_id": "use-case-mass-spectrum-molecule-shortlisting",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "msalign-2026-table3-task-massspecgym-formula-split-no-formula-r-1",
      "evaluation_ids": [
        "msalign-2026-table3-evaluation-deepsets-massspecgym-formula-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-emb-cos-massspecgym-formula-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-ffn-massspecgym-formula-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-jestr-massspecgym-formula-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-msalign-massspecgym-formula-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-sail-massspecgym-formula-split-no-formula-r-1"
      ],
      "endpoint": "Recall@1 of the true molecular structure among 256 mass-matched candidate structures on the MassSpecGym formula split, with no molecular formula supplied.",
      "relevance": "proxy",
      "rationale": "These source-checked retrieval evaluations inform method selection under explicit candidate and split assumptions; they do not establish accuracy in a new assay or unrestricted chemical search.",
      "constraints": [
        "Table 3, MassSpecGym formula split only; Recall@1. Retain all six formula-free configurations from that panel.",
        "Candidate sets contain 256 PubChem structures selected with a 10 ppm neutral-mass tolerance. The true molecular identity is assumed to be in the set.",
        "The formula data split groups by chemical formula; it does not supply formula to the evaluated models. Formula-conditioned methods and filtering are outside this mapping.",
        "These are the source paper implementations and training settings. Exact original-paper or current-release equivalence is not established.",
        "Metric groups at different shortlist sizes share an evaluation setting and must not count as independent replications."
      ],
      "limitations": [
        "The correct structure must be in a 256-candidate mass-filtered pool. Scores do not transfer automatically to a larger database, a different pool-construction procedure or a missing true structure.",
        "MCES and formula splits impose different distribution shifts. The source authors favour formula splits for their deployment interpretation; that choice does not prove that formula-split performance represents every intended application.",
        "Reported methods are the paper implementations: DeepSets training was extended to 50 epochs. These are not interchangeable with original model-paper scores or current checkpoints.",
        "Exact test denominators, split-manifest hashes, checkpoint hashes and evaluator revisions remain unextracted or unreported in the catalogue. No uncertainty values or new independent reproduction are established.",
        "Candidate recall is not an authenticated metabolite identification, a clinical diagnosis or a calibrated probability. Human domain review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "evidence-official-760ab2fa8c396aeb796c",
          "locator": "Table 3, MassSpecGym formula split block, row R@1, formula-free columns FFN, DeepSets, JESTR, Emb-Cos, SAIL and MSAlign; Sections 4, 5.1 and 5.2"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "264c270dc14e4c28740837553d6fcbf899529f22d829f825f065a798a4dfe76a"
    },
    {
      "id": "use-case-mapping-msms-formula-no-formula-r20",
      "use_case_id": "use-case-mass-spectrum-molecule-shortlisting",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "msalign-2026-table3-task-massspecgym-formula-split-no-formula-r-20",
      "evaluation_ids": [
        "msalign-2026-table3-evaluation-deepsets-massspecgym-formula-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-emb-cos-massspecgym-formula-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-ffn-massspecgym-formula-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-jestr-massspecgym-formula-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-msalign-massspecgym-formula-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-sail-massspecgym-formula-split-no-formula-r-20"
      ],
      "endpoint": "Recall@20 of the true molecular structure among 256 mass-matched candidate structures on the MassSpecGym formula split, with no molecular formula supplied.",
      "relevance": "proxy",
      "rationale": "These source-checked retrieval evaluations inform method selection under explicit candidate and split assumptions; they do not establish accuracy in a new assay or unrestricted chemical search.",
      "constraints": [
        "Table 3, MassSpecGym formula split only; Recall@20. Retain all six formula-free configurations from that panel.",
        "Candidate sets contain 256 PubChem structures selected with a 10 ppm neutral-mass tolerance. The true molecular identity is assumed to be in the set.",
        "The formula data split groups by chemical formula; it does not supply formula to the evaluated models. Formula-conditioned methods and filtering are outside this mapping.",
        "These are the source paper implementations and training settings. Exact original-paper or current-release equivalence is not established.",
        "Metric groups at different shortlist sizes share an evaluation setting and must not count as independent replications."
      ],
      "limitations": [
        "The correct structure must be in a 256-candidate mass-filtered pool. Scores do not transfer automatically to a larger database, a different pool-construction procedure or a missing true structure.",
        "MCES and formula splits impose different distribution shifts. The source authors favour formula splits for their deployment interpretation; that choice does not prove that formula-split performance represents every intended application.",
        "Reported methods are the paper implementations: DeepSets training was extended to 50 epochs. These are not interchangeable with original model-paper scores or current checkpoints.",
        "Exact test denominators, split-manifest hashes, checkpoint hashes and evaluator revisions remain unextracted or unreported in the catalogue. No uncertainty values or new independent reproduction are established.",
        "Candidate recall is not an authenticated metabolite identification, a clinical diagnosis or a calibrated probability. Human domain review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "evidence-official-760ab2fa8c396aeb796c",
          "locator": "Table 3, MassSpecGym formula split block, row R@20, formula-free columns FFN, DeepSets, JESTR, Emb-Cos, SAIL and MSAlign; Sections 4, 5.1 and 5.2"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "7b93e2000caca7115c4b176f38059899025b4efeb98f22a3e469bdc0c9521e1d"
    },
    {
      "id": "use-case-mapping-msms-mces-no-formula-r1",
      "use_case_id": "use-case-mass-spectrum-molecule-shortlisting",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "msalign-2026-table3-task-massspecgym-mces-split-no-formula-r-1",
      "evaluation_ids": [
        "msalign-2026-table3-evaluation-deepsets-massspecgym-mces-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-emb-cos-massspecgym-mces-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-ffn-massspecgym-mces-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-jestr-massspecgym-mces-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-msalign-massspecgym-mces-split-no-formula-r-1",
        "msalign-2026-table3-evaluation-sail-massspecgym-mces-split-no-formula-r-1"
      ],
      "endpoint": "Recall@1 of the true molecular structure among 256 mass-matched candidate structures on the MassSpecGym MCES split, with no molecular formula supplied.",
      "relevance": "proxy",
      "rationale": "These source-checked retrieval evaluations inform method selection under explicit candidate and split assumptions; they do not establish accuracy in a new assay or unrestricted chemical search.",
      "constraints": [
        "Table 3, MassSpecGym MCES split only; Recall@1. Retain all six formula-free configurations from that panel.",
        "Candidate sets contain 256 PubChem structures selected with a 10 ppm neutral-mass tolerance. The true molecular identity is assumed to be in the set.",
        "The MCES split separates structure clusters with a minimum MCES distance greater than 10. Molecular formula is not supplied to these models; formula-conditioned methods and filtering are outside this mapping.",
        "These are the source paper implementations and training settings. Exact original-paper or current-release equivalence is not established.",
        "Metric groups at different shortlist sizes share an evaluation setting and must not count as independent replications."
      ],
      "limitations": [
        "The correct structure must be in a 256-candidate mass-filtered pool. Scores do not transfer automatically to a larger database, a different pool-construction procedure or a missing true structure.",
        "MCES and formula splits impose different distribution shifts. The source authors favour formula splits for their deployment interpretation; that choice does not prove that formula-split performance represents every intended application.",
        "Reported methods are the paper implementations: DeepSets training was extended to 50 epochs. These are not interchangeable with original model-paper scores or current checkpoints.",
        "Exact test denominators, split-manifest hashes, checkpoint hashes and evaluator revisions remain unextracted or unreported in the catalogue. No uncertainty values or new independent reproduction are established.",
        "Candidate recall is not an authenticated metabolite identification, a clinical diagnosis or a calibrated probability. Human domain review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "evidence-official-760ab2fa8c396aeb796c",
          "locator": "Table 3, MassSpecGym MCES split block, row R@1, formula-free columns FFN, DeepSets, JESTR, Emb-Cos, SAIL and MSAlign; Sections 4, 5.1 and 5.2"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "1556c455cef29ff84665800c8a0bfd3eacef978dcfb6f0048ed261d55e4bd55c"
    },
    {
      "id": "use-case-mapping-msms-mces-no-formula-r20",
      "use_case_id": "use-case-mass-spectrum-molecule-shortlisting",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "msalign-2026-table3-task-massspecgym-mces-split-no-formula-r-20",
      "evaluation_ids": [
        "msalign-2026-table3-evaluation-deepsets-massspecgym-mces-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-emb-cos-massspecgym-mces-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-ffn-massspecgym-mces-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-jestr-massspecgym-mces-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-msalign-massspecgym-mces-split-no-formula-r-20",
        "msalign-2026-table3-evaluation-sail-massspecgym-mces-split-no-formula-r-20"
      ],
      "endpoint": "Recall@20 of the true molecular structure among 256 mass-matched candidate structures on the MassSpecGym MCES split, with no molecular formula supplied.",
      "relevance": "proxy",
      "rationale": "These source-checked retrieval evaluations inform method selection under explicit candidate and split assumptions; they do not establish accuracy in a new assay or unrestricted chemical search.",
      "constraints": [
        "Table 3, MassSpecGym MCES split only; Recall@20. Retain all six formula-free configurations from that panel.",
        "Candidate sets contain 256 PubChem structures selected with a 10 ppm neutral-mass tolerance. The true molecular identity is assumed to be in the set.",
        "The MCES split separates structure clusters with a minimum MCES distance greater than 10. Molecular formula is not supplied to these models; formula-conditioned methods and filtering are outside this mapping.",
        "These are the source paper implementations and training settings. Exact original-paper or current-release equivalence is not established.",
        "Metric groups at different shortlist sizes share an evaluation setting and must not count as independent replications."
      ],
      "limitations": [
        "The correct structure must be in a 256-candidate mass-filtered pool. Scores do not transfer automatically to a larger database, a different pool-construction procedure or a missing true structure.",
        "MCES and formula splits impose different distribution shifts. The source authors favour formula splits for their deployment interpretation; that choice does not prove that formula-split performance represents every intended application.",
        "Reported methods are the paper implementations: DeepSets training was extended to 50 epochs. These are not interchangeable with original model-paper scores or current checkpoints.",
        "Exact test denominators, split-manifest hashes, checkpoint hashes and evaluator revisions remain unextracted or unreported in the catalogue. No uncertainty values or new independent reproduction are established.",
        "Candidate recall is not an authenticated metabolite identification, a clinical diagnosis or a calibrated probability. Human domain review remains outstanding."
      ],
      "citations": [
        {
          "source_id": "evidence-official-760ab2fa8c396aeb796c",
          "locator": "Table 3, MassSpecGym MCES split block, row R@20, formula-free columns FFN, DeepSets, JESTR, Emb-Cos, SAIL and MSAlign; Sections 4, 5.1 and 5.2"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "20b4477c9d7da7d6213c216c7729530cbc41539830e44375d8be88fd41a66c32"
    },
    {
      "id": "use-case-mapping-plant-promoter-maize-protoplasts-a-thaliana",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-maize-protoplasts-a-thaliana",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-maize-protoplasts-maize-protoplasts-a-thaliana",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-maize-protoplasts-maize-protoplasts-a-thaliana"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Arabidopsis thaliana sequences in maize protoplasts, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Arabidopsis thaliana promoter reporter experiments in maize protoplasts. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Arabidopsis thaliana, measured in the maize protoplasts reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 2 (data row 1), column R2; Species=A. thaliana; Model=Maize model; Type=AgroNT; Figures/Fig3_panele.txt, line 8 (data row 7), column R2; Species=A. thaliana; Model=Maize model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "51291aa4704085356887b47aa8c1326b2527631d19b94aaa9422bd7aae65c73d"
    },
    {
      "id": "use-case-mapping-plant-promoter-maize-protoplasts-s-bicolor",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-maize-protoplasts-s-bicolor",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-maize-protoplasts-maize-protoplasts-s-bicolor",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-maize-protoplasts-maize-protoplasts-s-bicolor"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Sorghum bicolor sequences in maize protoplasts, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Sorghum bicolor promoter reporter experiments in maize protoplasts. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Sorghum bicolor, measured in the maize protoplasts reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 3 (data row 2), column R2; Species=S. bicolor; Model=Maize model; Type=AgroNT; Figures/Fig3_panele.txt, line 9 (data row 8), column R2; Species=S. bicolor; Model=Maize model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "5471e1de9dd6577dcea6493aa23ff5afc73249f06b6e80e63f0a4727db673254"
    },
    {
      "id": "use-case-mapping-plant-promoter-maize-protoplasts-z-mays",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-maize-protoplasts-z-mays",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-maize-protoplasts-maize-protoplasts-z-mays",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-maize-protoplasts-maize-protoplasts-z-mays"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Zea mays sequences in maize protoplasts, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Zea mays promoter reporter experiments in maize protoplasts. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Zea mays, measured in the maize protoplasts reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 4 (data row 3), column R2; Species=Z. mays; Model=Maize model; Type=AgroNT; Figures/Fig3_panele.txt, line 10 (data row 9), column R2; Species=Z. mays; Model=Maize model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "475520831178904bc0c433ecd74d16b6336f195ce222cee5a249c18d6a36c6bc"
    },
    {
      "id": "use-case-mapping-plant-promoter-tobacco-leaves-a-thaliana",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-tobacco-leaves-a-thaliana",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-tobacco-leaves-tobacco-leaves-a-thaliana",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-tobacco-leaves-tobacco-leaves-a-thaliana"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Arabidopsis thaliana sequences in tobacco leaves, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Arabidopsis thaliana promoter reporter experiments in tobacco leaves. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Arabidopsis thaliana, measured in the tobacco leaves reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 5 (data row 4), column R2; Species=A. thaliana; Model=Tobacco model; Type=AgroNT; Figures/Fig3_panele.txt, line 11 (data row 10), column R2; Species=A. thaliana; Model=Tobacco model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "0ff8f7e2efa02a37a5ea958b65bb39cecb4c3bd5ede6320efe03cb5ba6b7ab81"
    },
    {
      "id": "use-case-mapping-plant-promoter-tobacco-leaves-s-bicolor",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-tobacco-leaves-s-bicolor",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-tobacco-leaves-tobacco-leaves-s-bicolor",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-tobacco-leaves-tobacco-leaves-s-bicolor"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Sorghum bicolor sequences in tobacco leaves, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Sorghum bicolor promoter reporter experiments in tobacco leaves. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Sorghum bicolor, measured in the tobacco leaves reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 6 (data row 5), column R2; Species=S. bicolor; Model=Tobacco model; Type=AgroNT; Figures/Fig3_panele.txt, line 12 (data row 11), column R2; Species=S. bicolor; Model=Tobacco model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "133865d16861493f70b2c49fd1c97a40a92964a2ceb2358a8d22cc7b72838611"
    },
    {
      "id": "use-case-mapping-plant-promoter-tobacco-leaves-z-mays",
      "use_case_id": "use-case-plant-promoter-reporters",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "agront-2024-fig3e-task-tobacco-leaves-z-mays",
      "evaluation_ids": [
        "agront-2024-fig3e-evaluation-agront-promoter-strength-tobacco-leaves-tobacco-leaves-z-mays",
        "agront-2024-fig3e-evaluation-cnn-jores-et-al-promoter-strength-tobacco-leaves-tobacco-leaves-z-mays"
      ],
      "endpoint": "Predicting STARR-seq core-promoter strength for Zea mays sequences in tobacco leaves, assessed by held-out R².",
      "relevance": "proxy",
      "rationale": "This comparison can inform method selection for Zea mays promoter reporter experiments in tobacco leaves. Held-out promoter prediction remains a proxy for selecting newly designed candidates and requires validation in the intended experiment.",
      "constraints": [
        "170 bp core-promoter sequences from Zea mays, measured in the tobacco leaves reporter system; use this exact protocol and its two evaluated configurations.",
        "AgroNT uses task-specific regression with IA3 fine-tuning; the comparator is the task-specific CNN from Jores et al. Preserve assay-specific model identity.",
        "Compare R² only within this protocol. Do not pool assay hosts, sequence species or promoter and terminator tasks."
      ],
      "limitations": [
        "Author-reported comparison, source checked but not independently reproduced. No confidence intervals or run-variability estimate are supplied.",
        "Fitted checkpoint hashes, exact scoring counts, task-specific seeds and executable split manifests are unreported or unextracted.",
        "No prospective design success, endogenous expression, stable-plant, tissue-transfer or crop-yield claim. R² is not candidate hit rate."
      ],
      "citations": [
        {
          "source_id": "agront-2024-fig3e-source",
          "locator": "Figures/Fig3_panele.txt, line 7 (data row 6), column R2; Species=Z. mays; Model=Tobacco model; Type=AgroNT; Figures/Fig3_panele.txt, line 13 (data row 12), column R2; Species=Z. mays; Model=Tobacco model; Type=CNN (Jores et al.)"
        },
        {
          "source_id": "agront-2024-paper-methods",
          "locator": "Figure 3e and caption; Fine-tuning strategy (Sec16); Promoter and terminator strength prediction (Sec21)"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "2ae000a01a35c24d345e32e99953aebcc54e266459abadebd9875174f75804d6"
    },
    {
      "id": "use-case-mapping-protein-stability-amfr-esm2",
      "use_case_id": "use-case-protein-stability",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial bounded applicability review of the existing AMFR ESM-2 evaluation.",
      "protocol_id": "rewire-protocol-proteingym-amfr-v13",
      "evaluation_ids": [
        "rewire-local-20260920-evaluation-proteingym-esm2"
      ],
      "endpoint": "Ranking the mixed AMFR assay cohort by proteolysis-inferred folding stability using ESM-2 8M masked marginals.",
      "relevance": "proxy",
      "rationale": "This completed short-construct assay is a narrow example for assessing stability-ranking evidence. Its mixed single/double cohort does not directly answer the planned single-substitution comparison or establish transfer to another protein.",
      "constraints": [
        "One 47-residue AMFR construct and all 2,972 variants, comprising 820 singles and 2,152 doubles; no train/test split.",
        "Exact esm2_t6_8M_UR50D checkpoint, wild-type-context masked marginals summed over substituted sites, CPU, one thread; no assay labels, alignment or structure as scorer inputs.",
        "Protocol-only mapping: a direct reviewed task-membership relationship is not recorded."
      ],
      "limitations": [
        "No interval or seed-variability estimate; training overlap remains unknown and assay bytes have not been independently authenticated against an upstream checksum.",
        "Recorded inference timing excludes loading, preparation and metrics; peak memory is unreported.",
        "The site-independent and full EVmutation methods in the planned singles study have no completed comparison here.",
        "No general whole-protein, ProteinGym-wide or clinical inference; automated source review is not human scientific validation."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260920-source-proteingym-esm2",
          "locator": "/metrics; /coverage; /execution/adapter_provenance; /execution; /input_information; /provenance; /protocol_results"
        },
        {
          "source_id": "rewire-local-20260920-instructions-proteingym",
          "locator": "Data and method; Reproduce"
        },
        {
          "source_id": "profile-protocol-proteingym-reference-files-dms-substitutions-csv-a8f49801",
          "locator": "AMFR_HUMAN_Tsuboyama_2023_4G3O row: seq_len, selection_assay, raw_DMS_phenotype_name and mutant-count columns"
        },
        {
          "source_id": "use-case-source-amfr-pilot-plan-194a78b",
          "locator": "Question; Task and data; Methods; Gates and execution order; Runtime and cost"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-25T15:28:11Z",
        "note": "Reviewed existing measured evidence separately from the planned singles comparison. No new experiment, human domain review or clinical validation."
      },
      "evidence_sha256": "81883d91b243e8a9c958bca976ad951aa82ba5a130eff9263dc8f3a04707012e"
    },
    {
      "id": "use-case-mapping-protein-stability-amfr-random",
      "use_case_id": "use-case-protein-stability",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial review of the separate fixed-seed random control as limited context.",
      "protocol_id": "rewire-protocol-proteingym-amfr-random-v13",
      "evaluation_ids": [
        "rewire-local-20260921-evaluation-proteingym-random"
      ],
      "endpoint": "One fixed-seed random ranking of the mixed AMFR stability cohort.",
      "relevance": "proxy",
      "rationale": "A recorded null control helps inspect the assay and evaluation procedure. It is not a biological prediction method recommendation, a chance-performance interval or a matched comparison with the separately executed ESM-2 protocol.",
      "constraints": [
        "All 2,972 AMFR variants, mixing single and double substitutions; a single fixed seed of 0.",
        "SHA256 ranking of prepared variant IDs; no biological prediction or label fitting.",
        "Protocol-only mapping; keep its evaluation and configuration separate from ESM-2 and the planned singles comparison."
      ],
      "limitations": [
        "One seed does not estimate chance variability or uncertainty. Do not subtract these separate protocol results to assert an evaluated winner.",
        "Local input hashes do not independently authenticate assay bytes against an upstream published checksum.",
        "This control supplies no pathogenicity or clinical suitability evidence; no human scientific review or independent replication is recorded."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260921-source-proteingym-random",
          "locator": "/metrics; /coverage; /model_configuration; /protocol_configuration; /provenance; /protocol_results"
        },
        {
          "source_id": "rewire-local-20260921-instructions-proteingym-random",
          "locator": "Pinned reproduction instructions and execution scope"
        },
        {
          "source_id": "use-case-source-amfr-pilot-plan-194a78b",
          "locator": "Question; Task and data; Methods (N0); Analysis populations"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-25T15:28:11Z",
        "note": "Reviewed as a separate one-seed control, not clinical evidence or a cross-protocol comparison."
      },
      "evidence_sha256": "2fa413430c25c90a4184be20e096c2200185e566d1690953e29bc40d55c98ddf"
    },
    {
      "id": "use-case-mapping-rhodopsin-wavelength-rhomax-fixed-probes",
      "use_case_id": "use-case-rhodopsin-wavelength-transfer",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "rewire-protocol-flip2-rhomax-by-wild-type-v1",
      "evaluation_ids": [
        "rewire-local-20260920-evaluation-flip2-composition",
        "rewire-local-20260920-evaluation-flip2-train-mean",
        "rewire-local-20260921-evaluation-composition22",
        "rewire-local-20260921-evaluation-esm2-35m",
        "rewire-local-20260921-evaluation-esm2-8m"
      ],
      "endpoint": "Spearman rank association and full-ranking NDCG for absorption-wavelength predictions on 184 held-out Rhomax sequences.",
      "relevance": "proxy",
      "rationale": "The archived background split directly tests one transfer setting; choosing constructs in a new experiment requires validation beyond those 184 rows.",
      "constraints": [
        "The same frozen Rhomax by_wild_type assignments cover 584 training, 116 validation and 184 test records. Validation labels do not select these fixed settings.",
        "Two exact ESM-2 checkpoints use frozen final-layer residue means, no MSA or templates, and a train-only alpha-10 ridge probe with train-only target scaling. They are not zero-shot likelihood scorers.",
        "The 40-feature and 22-feature composition controls are distinct configurations. Training mean is the existing constant-control evaluation; repeated control checks do not add replications."
      ],
      "limitations": [
        "A spectral-tuning endpoint is not opsin activation efficiency, expression, photostability, cellular function, general protein fitness or clinical usefulness.",
        "Spearman and full-ranking NDCG do not establish wavelength calibration, top-k precision or a prospective experimental hit rate. Constant predictions have undefined Spearman; their NDCG is a control value, not strong predictive evidence.",
        "No intervals or seed-variability estimates. The 35M checkpoint was chosen after seeing the 8M outcome; the source describes an exploratory follow-up, not a preregistered family-scale comparison.",
        "ESM-2 pretraining overlap is unresolved. A frozen linear probe result does not establish the value of alternative pooling, fine-tuning or every model-family configuration.",
        "Recorded timing covers sections of execution and is not a hardware-normalised deployment-cost comparison. Human domain review and external replication are not established."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260921-instructions-esm2-35m",
          "locator": "Matched Rhomax evaluations; Procedure and evidence; Re-run or inspect"
        },
        {
          "source_id": "rewire-local-20260921-source-esm2-35m",
          "locator": "/metrics; /coverage; /model_configuration; /protocol_configuration; /execution; /provenance"
        },
        {
          "source_id": "rewire-local-20260921-source-esm2-8m",
          "locator": "/metrics; /coverage; /model_configuration; /protocol_configuration; /execution; /provenance"
        },
        {
          "source_id": "rewire-local-20260921-source-composition22",
          "locator": "/metrics; /model_configuration; /protocol_configuration"
        },
        {
          "source_id": "rewire-local-20260920-source-flip2-train-mean",
          "locator": "/metrics; /coverage; /model_configuration"
        },
        {
          "source_id": "rewire-local-20260920-source-flip2-composition",
          "locator": "/metrics; /coverage; /model_configuration; /provenance"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:11:40.845Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "a012ef5867a510fe9861d1a3de019095e38fb8718f0dbd3581ddd1c5b9a51783"
    },
    {
      "id": "use-case-mapping-splicing-mfass-matched-v1",
      "use_case_id": "use-case-splicing-follow-up",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial bounded applicability review of the matched MFASS study.",
      "protocol_id": "rewire-mfass-matched-v1-protocol",
      "task_id": "catalog-task-mfass-splice",
      "evaluation_ids": [
        "rewire-mfass-matched-v1-evaluation-s0",
        "rewire-mfass-matched-v1-evaluation-s1",
        "rewire-mfass-matched-v1-evaluation-p0",
        "rewire-mfass-matched-v1-evaluation-p1"
      ],
      "endpoint": "Ranking held-out MFASS SNVs by reporter-assay splice disruption under matched canonical annotation.",
      "relevance": "proxy",
      "rationale": "The recorded endpoint informs assay-oriented prioritisation under its declared conditions. Selecting variants for a different follow-up experiment requires transfer validation; the endpoint is not patient RNA or clinical pathogenicity.",
      "constraints": [
        "Identical scored population: 8,297 of 8,324 held-out variants, 314 scored positives and 460 groups in all four configurations.",
        "Shared GENCODE 44 canonical transcript selection and FASTA; 50-base distance; distinct SpliceAI and Pangolin masking settings.",
        "Keep historical annotation conditions and scoring populations in their existing separate comparison groups."
      ],
      "limitations": [
        "23 assembly-orientation mismatches and four canonical-transcript-span exclusions remain unscored. The latter are protocol exclusions, not established faulty variants; full-population performance is unknown.",
        "No top-100 precision difference is established and masked Pangolin is tie-sensitive. Paired contrast intervals are not individual-condition intervals.",
        "Exploratory source review only; no independent human review, replication or clinical validation. Assembly-issue author confirmation is not established."
      ],
      "citations": [
        {
          "source_id": "rewire-mfass-matched-v1-source-report",
          "locator": "/conditions/{S0,S1,P0,P1}/coverage; /conditions/{S0,S1,P0,P1}/metrics; /conditions/{S0,S1,P0,P1}/ties; /contrasts/{S1-S0,P1-P0,P0-S0}/paired/precision_at_capacity"
        },
        {
          "source_id": "rewire-mfass-matched-v1-source-manifest-v1",
          "locator": "/conditions; /distance; /resources_sha256; /code"
        },
        {
          "source_id": "rewire-mfass-matched-v1-source-exclusion-verification",
          "locator": "/exclusion_counts; /conditions; /checks"
        },
        {
          "source_id": "use-case-source-mfass-matched-intake-194a78b",
          "locator": "Opening coverage, exclusion, annotation and interpretation paragraphs; Review and validation"
        },
        {
          "source_id": "evidence-expansion-mfass-readme-62a93814",
          "locator": "Dataset; Limits"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-25T15:28:11Z",
        "note": "Reviewed exact configuration and population scope against pinned source bytes. Applicability remains proxy evidence; no human domain review or clinical validation."
      },
      "evidence_sha256": "c4cfbe66c30d59308fc8d2226f42720a09a6e54d2f407c7141b08e01e39c0a47"
    },
    {
      "id": "use-case-mapping-utr-translation-baselines-designed-v1",
      "use_case_id": "use-case-utr-translation-baselines",
      "lifecycle": "active",
      "revision": 1,
      "reason": "Initial applicability review: primary sources and separate automated cross-review support this exact protocol, its complete selected comparator group and the stated limits.",
      "protocol_id": "rewire-protocol-mrnabench-designed-mrl-v1",
      "evaluation_ids": [
        "rewire-local-20260920-evaluation-mrnabench-composition",
        "rewire-local-20260920-evaluation-mrnabench-train-mean"
      ],
      "endpoint": "Predicting target_mrl_designed on the complete 15,003-record canonical test split of the Sample designed reporter dataset.",
      "relevance": "direct",
      "rationale": "The matched sequence-composition and training-mean controls provide an auditable starting point for deciding whether a more complex method adds predictive value on this defined endpoint. They do not establish transfer to a different library or supply a pretrained-model comparison.",
      "constraints": [
        "Both controls use the same full processed sequences, seed-2541 split, training labels and complete held-out test population. No validation or test labels select hyperparameters.",
        "Composition features are log1p length, A/C/G/T fractions and unknown fraction, fitted with train-only RidgeCV over the recorded five alpha values and no normalization. The constant control is the training-target arithmetic mean.",
        "MSE is the matched primary endpoint. Constant-control Pearson and Spearman are unavailable; do not replace missing values with zeros."
      ],
      "limitations": [
        "One target, one split and one execution per method; no uncertainty or seed variability. No homology-separated or compositional-generalisation claim.",
        "This training-only protocol differs from upstream probing and is not an aggregate mRNABench score or a reproduction of a published model result.",
        "The original run’s raw inputs and saved predictions are not publicly archived. The recipe obtains source data and runs the controls again, including separate protein controls; it does not rescore prior predictions. Source-data reuse terms and cross-platform reproducibility remain unresolved.",
        "Research baseline evidence only, with no new execution, human domain review or clinical validation."
      ],
      "citations": [
        {
          "source_id": "rewire-local-20260920-source-mrnabench-composition",
          "locator": "/coverage; /model_configuration; /protocol_configuration; /protocol_results; /provenance"
        },
        {
          "source_id": "rewire-local-20260920-source-mrnabench-train-mean",
          "locator": "/coverage; /model_configuration; /protocol_results/metric_unavailable_reasons; /provenance"
        },
        {
          "source_id": "rewire-local-20260920-instructions-mrnabench",
          "locator": "Results: RNA evaluation paragraph; Provenance and review; Reproduce the four sequence controls"
        }
      ],
      "review": {
        "method": "automated_source_review",
        "actor": "Codex research curation",
        "reviewed_at": "2026-09-28T12:15:35.948Z",
        "note": "Primary-source curation and separate automated cross-review checked exact evidence identities, comparator coverage, endpoint relevance and transfer limits. No new execution, human scientific review, independent replication or clinical validation."
      },
      "evidence_sha256": "1034171098b5ead5f29764df1c6efaccee4aa4c24f235945e36b8536ec30caf5"
    }
  ],
  "release_id": "2026-09-28-c7b5ac6d34f2",
  "input_sha256": "a0dd27a5f430ec387d309fd6e8615083250873ce4cff5e882c6fb8458ef80e95"
}
