{
  "schema_version": "1.0",
  "release_id": "2026-09-29-06401fd5b220",
  "items": [
    {
      "schema_version": "1.0",
      "id": "mfass-v2-discrepancies",
      "title": "MFASS v2: model disagreement",
      "question": "Does model disagreement depend on exon boundary distance, assay replicate agreement or gene/exon concentration?",
      "catalogue_release_id": "2026-09-20-370b30415b09",
      "dataset_id": "rewire-mfass-v2-dataset",
      "evaluation_ids": [
        "rewire-evaluation-baseline-kmer-position-v2",
        "rewire-evaluation-dnabert2-117m-frozen-pair-logreg",
        "rewire-evaluation-spliceai-1-3-1",
        "rewire-evaluation-pangolin-maskfalse"
      ],
      "protocol_id": "rewire-mfass-v2",
      "sdk_protocol_id": "mfass-v2",
      "runner_code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
      "artifacts": [
        {
          "id": "mfass-v2-discrepancies-cohort",
          "role": "cohort",
          "sha256": "389702ff4c647d7ce10a90092a6fa811ae777d15997baf39ce9aae0346247bd0",
          "format": "tsv",
          "uri": null
        },
        {
          "id": "mfass-v2-discrepancies-outcomes",
          "role": "outcomes",
          "sha256": "a637ca0e307e66ff48811ec7efa22b9ce453bc7883b04f0cacb867f7283132d8",
          "format": "txt",
          "uri": "https://raw.githubusercontent.com/KosuriLab/MFASS/master/processed_data/snv/snv_data_clean.txt"
        },
        {
          "id": "mfass-v2-discrepancies-annotations",
          "role": "annotations",
          "sha256": "71a857fe647c4e68acbb41ca61e959c47e1176de89b1442bd6ca1772aa60d5a1",
          "format": "txt",
          "uri": "https://raw.githubusercontent.com/KosuriLab/MFASS/master/processed_data/snv/snv_func_annot.txt"
        },
        {
          "id": "mfass-v2-discrepancies-split",
          "role": "split",
          "sha256": "999ebcb7e63a5c5eaa8780fa468e59ac1f934260ad50102814174c396317f052",
          "format": "tsv",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/splits/split-v2.tsv"
        },
        {
          "id": "mfass-v2-discrepancies-baseline-kmer-position-v2-scores",
          "role": "baseline-kmer-position-v2-scores",
          "sha256": "ed0ebb3d74183deb1fb475d0e706c0b1211fd6ab9d20449c2c194e57f912d862",
          "format": "npy",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/baseline-kmer-position-v2.scores.npy"
        },
        {
          "id": "mfass-v2-discrepancies-baseline-kmer-position-v2-predictions",
          "role": "baseline-kmer-position-v2-predictions",
          "sha256": "2f3117c225a8da9ea737abfa2d0f7e1dec97696ae4bacd263864bec9d7f923eb",
          "format": "tsv",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/baseline-kmer-position-v2.predictions.tsv"
        },
        {
          "id": "mfass-v2-discrepancies-baseline-kmer-position-v2-report",
          "role": "baseline-kmer-position-v2-report",
          "sha256": "9a0b78674cc714177fec6e8c487588d6186d4fe93d6d7dbcb48bed6858885e15",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/baseline-kmer-position-v2.json"
        },
        {
          "id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-scores",
          "role": "dnabert2-117m-frozen-pair-logreg-scores",
          "sha256": "e6b019babf0b3fb9ec05d4261a9383c0c896183a76c3b31634564c7a63d01105",
          "format": "npy",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.scores.npy"
        },
        {
          "id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-predictions",
          "role": "dnabert2-117m-frozen-pair-logreg-predictions",
          "sha256": "3abded2932366e6e2c8d6e0da5836c4240239372758b3c68cd5f19635b8cee11",
          "format": "tsv",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.predictions.tsv"
        },
        {
          "id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-report",
          "role": "dnabert2-117m-frozen-pair-logreg-report",
          "sha256": "60b28541853de349f878d6d1ccbcbe7dfd21b69db976e9c01f60e88bdba0f2a1",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.json"
        },
        {
          "id": "mfass-v2-discrepancies-spliceai-1.3.1-predictions",
          "role": "spliceai-1.3.1-predictions",
          "sha256": "b39df773067e4ecd862980f177da69d6117c75c93c754f4cbdd7a77bd419851f",
          "format": "tsv",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/spliceai-1.3.1.predictions.tsv"
        },
        {
          "id": "mfass-v2-discrepancies-spliceai-1.3.1-report",
          "role": "spliceai-1.3.1-report",
          "sha256": "6d5c59eb0fd60d95064d97e331c9cdb12b2a34db7dc509ff73dfb9502ddf02dd",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/spliceai-1.3.1.json"
        },
        {
          "id": "mfass-v2-discrepancies-pangolin-maskFalse-predictions",
          "role": "pangolin-maskFalse-predictions",
          "sha256": "faa7d4cb3728111f1c5a8e8be6cd2c5958c4cced7c7a366ec96171d03b855ad6",
          "format": "tsv",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/pangolin-maskFalse.predictions.tsv"
        },
        {
          "id": "mfass-v2-discrepancies-pangolin-maskFalse-report",
          "role": "pangolin-maskFalse-report",
          "sha256": "bb0bb6732808699e54938233df1835dfc1f775f33ba7d6acd916e53d588a3c44",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/pangolin-maskFalse.json"
        },
        {
          "id": "mfass-v2-discrepancies-table",
          "role": "table",
          "sha256": "c57f9cfe91d02ab54f6bc20934f5b8f678b37c3b13d8d20aa805fa7d73e62c4f",
          "format": "json",
          "uri": null
        },
        {
          "id": "mfass-v2-discrepancies-receipt",
          "role": "receipt",
          "sha256": "a9fa54bcf99457f9b23424ad656a952eb262b376845327e267dc035cf85a012e",
          "format": "json",
          "uri": null
        },
        {
          "id": "mfass-v2-discrepancies-preparation_code",
          "role": "preparation_code",
          "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
          "format": "py",
          "uri": null
        },
        {
          "id": "mfass-v2-discrepancies-environment",
          "role": "environment",
          "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
          "format": "lock",
          "uri": null
        }
      ],
      "table_artifact_id": "mfass-v2-discrepancies-table",
      "semantics": {
        "target": "binary",
        "outcome": "assay splice disruption",
        "unit": "binary label",
        "score_direction": "higher",
        "join_key": "id",
        "independent_unit": "connected exon/gene group",
        "subgroup_fields": [
          "boundary_band",
          "replicate_gap_band",
          "strand",
          "legacy_orientation"
        ],
        "exposed": true,
        "split": "test"
      },
      "expected_metrics": {
        "baseline-kmer-position-v2": {
          "precision_at_capacity": 0.61,
          "recall_at_capacity": 0.19365079365079366,
          "average_precision_sklearn": 0.2864167459589237,
          "auroc": 0.7779498064677238
        },
        "dnabert2-117m-frozen-pair-logreg": {
          "precision_at_capacity": 0.03,
          "recall_at_capacity": 0.009523809523809525,
          "average_precision_sklearn": 0.04508654312652131,
          "auroc": 0.5500324040216661
        },
        "spliceai-1.3.1": {
          "precision_at_capacity": 0.64,
          "recall_at_capacity": 0.2077922077922078,
          "average_precision_sklearn": 0.2986855472760137,
          "auroc": 0.8055241740253153
        },
        "pangolin-maskFalse": {
          "precision_at_capacity": 0.65,
          "recall_at_capacity": 0.2070063694267516,
          "average_precision_sklearn": 0.3887617543064248,
          "auroc": 0.8756851300560864
        }
      },
      "metric_tolerance": 1e-9,
      "verification": {
        "verified_at": "2026-09-23T23:00:14.349323+00:00",
        "checks": [
          {
            "check": "artifact_hashes",
            "status": "passed",
            "detail": "Exact local bytes recorded; source pins checked where available."
          },
          {
            "check": "join_integrity",
            "status": "passed",
            "detail": "8324 unique test IDs; scores and outcomes reconcile."
          },
          {
            "check": "score_semantics",
            "status": "passed",
            "detail": "Higher scores predict higher assay splice disruption; metrics retain their own direction."
          },
          {
            "check": "metric_replay",
            "status": "passed",
            "detail": "Saved predictions reproduce recorded metrics within 1e-9 absolute tolerance."
          },
          {
            "check": "annotations",
            "status": "passed",
            "detail": "Registered subgroup fields derive from the preserved source snapshot."
          },
          {
            "check": "dependence",
            "status": "passed",
            "detail": "connected exon/gene group"
          }
        ],
        "limitations": [
          "Existing test outcomes have been inspected; all new subgroup findings are exploratory.",
          "Specialists use different genomic context and annotation releases; model-only attribution is unsupported.",
          "Boundary bands are symmetric distances, not canonical dinucleotide annotations; baseline uses distance features.",
          "The historical v1 sequence-orientation error is already corrected and is not a new discovery.",
          "Source assay data have no declared redistribution licence; obtain original tables from KosuriLab/MFASS."
        ]
      },
      "local_recipes": []
    },
    {
      "schema_version": "1.0",
      "id": "proteingym-amfr-discrepancies",
      "title": "ProteinGym AMFR: negative correlation",
      "question": "Does the negative association differ between single and multiple substitutions?",
      "catalogue_release_id": "2026-09-20-370b30415b09",
      "dataset_id": "rewire-dataset-proteingym-amfr-v13",
      "evaluation_ids": [
        "rewire-local-20260920-evaluation-proteingym-esm2"
      ],
      "protocol_id": "rewire-protocol-proteingym-amfr-v13",
      "sdk_protocol_id": "proteingym-v1.3-dms-substitutions",
      "runner_code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
      "artifacts": [
        {
          "id": "proteingym-amfr-discrepancies-prepared",
          "role": "prepared",
          "sha256": "d9e220a0e2b0675e73f4b79b4ae45b8f807dedbb886bbd4b2f8e886822ddf57f",
          "format": "json",
          "uri": null,
          "semantic_sha256": "9f0e0cce34942797364990ef4dc9fb8b382f7b2fd59e6c72de23efc2992fb148"
        },
        {
          "id": "proteingym-amfr-discrepancies-proteingym-esm2-predictions",
          "role": "proteingym-esm2-predictions",
          "sha256": "7ab0c8901406d5c1ab759038c8beef235721938394cfdf192c46671e98e0bbca",
          "format": "json",
          "uri": null
        },
        {
          "id": "proteingym-amfr-discrepancies-proteingym-esm2-report",
          "role": "proteingym-esm2-report",
          "sha256": "6a14b3866141d5779c76da8071d85c73b7cce0dc4e101c93122706873e847d26",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/proteingym-esm2/report.json"
        },
        {
          "id": "proteingym-amfr-discrepancies-proteingym-esm2-source",
          "role": "proteingym-esm2-source",
          "sha256": "90888f1f18625cdc2e9fb0bf7ecc1b1c4c2d6b700e8b4971e4349297926f996f",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/proteingym-esm2/retrieval.json"
        },
        {
          "id": "proteingym-amfr-discrepancies-table",
          "role": "table",
          "sha256": "4fac11aacc0f00a70e385ed7b8d73427aaa0716906a74e93bc3d48df5ff43f9f",
          "format": "json",
          "uri": null
        },
        {
          "id": "proteingym-amfr-discrepancies-receipt",
          "role": "receipt",
          "sha256": "9f117c65a3c860b83735d4874d06a807ee1a1523464e9afca493e7dcd9e59b56",
          "format": "json",
          "uri": null
        },
        {
          "id": "proteingym-amfr-discrepancies-preparation_code",
          "role": "preparation_code",
          "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
          "format": "py",
          "uri": null
        },
        {
          "id": "proteingym-amfr-discrepancies-environment",
          "role": "environment",
          "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
          "format": "lock",
          "uri": null
        }
      ],
      "table_artifact_id": "proteingym-amfr-discrepancies-table",
      "semantics": {
        "target": "continuous",
        "outcome": "DMS stability",
        "unit": "assay DMS score",
        "score_direction": "higher",
        "join_key": "id",
        "independent_unit": null,
        "subgroup_fields": [
          "mutation_class"
        ],
        "exposed": true,
        "split": "test"
      },
      "expected_metrics": {
        "proteingym-esm2": {
          "AUC": 0.394,
          "MCC": -0.139,
          "NDCG": 0.44,
          "Spearman": -0.209,
          "Top_recall": 0.057
        }
      },
      "metric_tolerance": 1e-9,
      "verification": {
        "verified_at": "2026-09-23T23:00:14.349323+00:00",
        "checks": [
          {
            "check": "artifact_hashes",
            "status": "passed",
            "detail": "Exact local bytes recorded; source pins checked where available."
          },
          {
            "check": "join_integrity",
            "status": "passed",
            "detail": "2972 unique test IDs; scores and outcomes reconcile."
          },
          {
            "check": "score_semantics",
            "status": "passed",
            "detail": "Higher scores predict higher DMS stability; metrics retain their own direction."
          },
          {
            "check": "metric_replay",
            "status": "passed",
            "detail": "Saved predictions reproduce recorded metrics within 1e-9 absolute tolerance."
          },
          {
            "check": "annotations",
            "status": "passed",
            "detail": "Registered subgroup fields derive from the preserved source snapshot."
          },
          {
            "check": "dependence",
            "status": "passed",
            "detail": "Independent sampling unit unknown; descriptive analysis only."
          }
        ],
        "limitations": [
          "Existing test outcomes are exposed; no independent validation is claimed.",
          "Independent experimental grouping is unavailable; subgroup summaries are descriptive.",
          "Prepared opaque IDs must not be joined to a newly prepared snapshot by ID.",
          "One protein construct, complete selected assay but partial benchmark suite.",
          "Official three-decimal metrics and score sign are preserved; no post-hoc sign reversal.",
          "Upstream archive checksums were observed locally, not independently source-verified."
        ]
      },
      "local_recipes": []
    },
    {
      "schema_version": "1.0",
      "id": "mrnabench-discrepancies",
      "title": "mRNABench designed: composition effects",
      "question": "Where does the composition predictor improve over the training-mean control?",
      "catalogue_release_id": "2026-09-20-370b30415b09",
      "dataset_id": "rewire-dataset-mrnabench-designed-mrl-v1",
      "evaluation_ids": [
        "rewire-local-20260920-evaluation-mrnabench-composition",
        "rewire-local-20260920-evaluation-mrnabench-train-mean"
      ],
      "protocol_id": "rewire-protocol-mrnabench-designed-mrl-v1",
      "sdk_protocol_id": "mrnabench-sample-mrl-v1",
      "runner_code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
      "artifacts": [
        {
          "id": "mrnabench-discrepancies-prepared",
          "role": "prepared",
          "sha256": "e5d6fb9e5fadec1f9e6235ca3ed2054ac05f476ccc8f5dc67ec500febeae8237",
          "format": "json",
          "uri": null,
          "semantic_sha256": "f4f9777b5119b739437c6a9d85e4e1b937268dafd1307efca5f072d38957738b"
        },
        {
          "id": "mrnabench-discrepancies-mrnabench-composition-predictions",
          "role": "mrnabench-composition-predictions",
          "sha256": "a8f80cfea7ddd197ef9446e161f5a761741b3f778dfc1459275c3355432e15c5",
          "format": "json",
          "uri": null
        },
        {
          "id": "mrnabench-discrepancies-mrnabench-composition-report",
          "role": "mrnabench-composition-report",
          "sha256": "8b68dfe5725bf5d8c108265039e3787f4b3b55986eab8a430082c4d446a4f960",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-composition/report.json"
        },
        {
          "id": "mrnabench-discrepancies-mrnabench-composition-source",
          "role": "mrnabench-composition-source",
          "sha256": "ef486f7be79a6f5589afd7494b7acfa13ca698da50f0ffd0e50fd6eba1eb9934",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-composition/source.json"
        },
        {
          "id": "mrnabench-discrepancies-mrnabench-train-mean-predictions",
          "role": "mrnabench-train-mean-predictions",
          "sha256": "fb435ffda2e56792a9c781fba8dfda535b3d6508cf28f484162c5fd9065a87d8",
          "format": "json",
          "uri": null
        },
        {
          "id": "mrnabench-discrepancies-mrnabench-train-mean-report",
          "role": "mrnabench-train-mean-report",
          "sha256": "c036d9fa75b6ba4d5c65d31cd575ee2bfc357beec3ca7e6c80f07e3128a04e71",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-train-mean/report.json"
        },
        {
          "id": "mrnabench-discrepancies-mrnabench-train-mean-source",
          "role": "mrnabench-train-mean-source",
          "sha256": "b8bbc4f3f2ef54d92721015205bdd73a81e4cfcbc596985e8724be3416899d54",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-train-mean/source.json"
        },
        {
          "id": "mrnabench-discrepancies-recipe_code",
          "role": "recipe_code",
          "sha256": "b3407940ca4c8a1cdcf3a3bb3849c3e26d7f85272b2ee63087307378ffaa79dc",
          "format": "py",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/packages/rewirebench/src/rewirebench/adapters/sequence.py"
        },
        {
          "id": "mrnabench-discrepancies-table",
          "role": "table",
          "sha256": "d49ff1b4dacdfe0663c35bdad6a9ad91e79f5862fe96f36a627acb8407610bf3",
          "format": "json",
          "uri": null
        },
        {
          "id": "mrnabench-discrepancies-receipt",
          "role": "receipt",
          "sha256": "5edeb10e42be225f6b85481c95eb8985ab1724d3af353737fdafd29ccdfc4022",
          "format": "json",
          "uri": null
        },
        {
          "id": "mrnabench-discrepancies-preparation_code",
          "role": "preparation_code",
          "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
          "format": "py",
          "uri": null
        },
        {
          "id": "mrnabench-discrepancies-environment",
          "role": "environment",
          "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
          "format": "lock",
          "uri": null
        }
      ],
      "table_artifact_id": "mrnabench-discrepancies-table",
      "semantics": {
        "target": "continuous",
        "outcome": "mean ribosome load",
        "unit": "MRL",
        "score_direction": "higher",
        "join_key": "id",
        "independent_unit": null,
        "subgroup_fields": [
          "duplicate_sequence",
          "length_band",
          "gc_band"
        ],
        "exposed": true,
        "split": "test"
      },
      "expected_metrics": {
        "mrnabench-composition": {
          "mse": 1.914439715839851,
          "pearson": 0.4367869170170987,
          "spearman": 0.49477537290902324
        },
        "mrnabench-train-mean": {
          "mse": 2.3655294722554605,
          "pearson": null,
          "spearman": null
        }
      },
      "metric_tolerance": 1e-9,
      "verification": {
        "verified_at": "2026-09-23T23:00:14.349323+00:00",
        "checks": [
          {
            "check": "artifact_hashes",
            "status": "passed",
            "detail": "Exact local bytes recorded; source pins checked where available."
          },
          {
            "check": "join_integrity",
            "status": "passed",
            "detail": "15003 unique test IDs; scores and outcomes reconcile."
          },
          {
            "check": "score_semantics",
            "status": "passed",
            "detail": "Higher scores predict higher mean ribosome load; metrics retain their own direction."
          },
          {
            "check": "metric_replay",
            "status": "passed",
            "detail": "Saved predictions reproduce recorded metrics within 1e-9 absolute tolerance."
          },
          {
            "check": "annotations",
            "status": "passed",
            "detail": "Registered subgroup fields derive from the preserved source snapshot."
          },
          {
            "check": "dependence",
            "status": "passed",
            "detail": "Independent sampling unit unknown; descriptive analysis only."
          },
          {
            "check": "recipe_pinned",
            "status": "passed",
            "detail": "Registered SDK recipe; implementation artifact and environment hashes retained."
          },
          {
            "check": "resource_estimate",
            "status": "passed",
            "detail": "Local CPU controls previously executed; all jobs remain subject to the campaign watchdog."
          }
        ],
        "limitations": [
          "Existing test outcomes are exposed; no independent validation is claimed.",
          "Independent experimental grouping is unavailable; subgroup summaries are descriptive.",
          "Prepared opaque IDs must not be joined to a newly prepared snapshot by ID.",
          "Random split does not establish homology separation.",
          "Source data reuse licence is unreported; no raw data redistribution."
        ]
      },
      "local_recipes": [
        "sdk:train-mean-v1",
        "sdk:sequence-composition-v1"
      ]
    },
    {
      "schema_version": "1.0",
      "id": "flip2-discrepancies",
      "title": "FLIP2 Rhomax: metric interpretation control",
      "question": "Why can a constant predictor have high NDCG and undefined Spearman correlation?",
      "catalogue_release_id": "2026-09-20-370b30415b09",
      "dataset_id": "rewire-dataset-flip2-rhomax-by-wild-type-v3",
      "evaluation_ids": [
        "rewire-local-20260920-evaluation-flip2-composition",
        "rewire-local-20260920-evaluation-flip2-train-mean"
      ],
      "protocol_id": "rewire-protocol-flip2-rhomax-by-wild-type-v1",
      "sdk_protocol_id": "flip2-fitness-v1",
      "runner_code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
      "artifacts": [
        {
          "id": "flip2-discrepancies-prepared",
          "role": "prepared",
          "sha256": "df996aee84bd5ed3e6381f9f6db864a377ab0b4809a9a48ff146d06a3dc069df",
          "format": "json",
          "uri": null,
          "semantic_sha256": "f417d43497e0c8c453bd7c66cfea05959b4d6d47462283096e42a2dae8b72419"
        },
        {
          "id": "flip2-discrepancies-flip2-composition-predictions",
          "role": "flip2-composition-predictions",
          "sha256": "72ea3c506b8645ae6020207f904fbd5c0b09549cd125fcdcabfb487e68681b9c",
          "format": "json",
          "uri": null
        },
        {
          "id": "flip2-discrepancies-flip2-composition-report",
          "role": "flip2-composition-report",
          "sha256": "abd08d3f493428d06f43740c94e268575d755560e72120c73b294127acb395e6",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-composition/report.json"
        },
        {
          "id": "flip2-discrepancies-flip2-composition-source",
          "role": "flip2-composition-source",
          "sha256": "7157313eab75ff8f93b934ac186852c8c82f8bb885617cab70c89427c631c1e4",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-composition/source.json"
        },
        {
          "id": "flip2-discrepancies-flip2-train-mean-predictions",
          "role": "flip2-train-mean-predictions",
          "sha256": "599dca4e6fa643a7cdfc96ac339033f61e5846ce225af17ac2ac3917ecc14fac",
          "format": "json",
          "uri": null
        },
        {
          "id": "flip2-discrepancies-flip2-train-mean-report",
          "role": "flip2-train-mean-report",
          "sha256": "eb89260f6e138bb9ae2fdf3b882a87fdf8aeabd54a0ac6f17a73f5671c96b4d8",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-train-mean/report.json"
        },
        {
          "id": "flip2-discrepancies-flip2-train-mean-source",
          "role": "flip2-train-mean-source",
          "sha256": "8bd712289f136edbfde3fa885166ef85e4f6fe3c4c437810c24dadb87107cc3f",
          "format": "json",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-train-mean/source.json"
        },
        {
          "id": "flip2-discrepancies-recipe_code",
          "role": "recipe_code",
          "sha256": "b3407940ca4c8a1cdcf3a3bb3849c3e26d7f85272b2ee63087307378ffaa79dc",
          "format": "py",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/packages/rewirebench/src/rewirebench/adapters/sequence.py"
        },
        {
          "id": "flip2-discrepancies-table",
          "role": "table",
          "sha256": "8562d4891270a73026fc7852c05f41206ad52e8fcae1d706d83ec57572c7c949",
          "format": "json",
          "uri": null
        },
        {
          "id": "flip2-discrepancies-receipt",
          "role": "receipt",
          "sha256": "68631e653cd5aa9047b3f4e88cbb3fec94630847c1b73fad8de1018423d81fac",
          "format": "json",
          "uri": null
        },
        {
          "id": "flip2-discrepancies-preparation_code",
          "role": "preparation_code",
          "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
          "format": "py",
          "uri": null
        },
        {
          "id": "flip2-discrepancies-environment",
          "role": "environment",
          "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
          "format": "lock",
          "uri": null
        }
      ],
      "table_artifact_id": "flip2-discrepancies-table",
      "semantics": {
        "target": "continuous",
        "outcome": "spectral wavelength",
        "unit": "nm",
        "score_direction": "higher",
        "join_key": "id",
        "independent_unit": null,
        "subgroup_fields": [
          "duplicate_sequence",
          "length_band"
        ],
        "exposed": true,
        "split": "test"
      },
      "expected_metrics": {
        "flip2-composition": {
          "ndcg": 0.954819941821582,
          "spearman": 0.41818225155801514
        },
        "flip2-train-mean": {
          "ndcg": 0.9206667522227658,
          "spearman": null
        }
      },
      "metric_tolerance": 1e-9,
      "verification": {
        "verified_at": "2026-09-23T23:00:14.349323+00:00",
        "checks": [
          {
            "check": "artifact_hashes",
            "status": "passed",
            "detail": "Exact local bytes recorded; source pins checked where available."
          },
          {
            "check": "join_integrity",
            "status": "passed",
            "detail": "184 unique test IDs; scores and outcomes reconcile."
          },
          {
            "check": "score_semantics",
            "status": "passed",
            "detail": "Higher scores predict higher spectral wavelength; metrics retain their own direction."
          },
          {
            "check": "metric_replay",
            "status": "passed",
            "detail": "Saved predictions reproduce recorded metrics within 1e-9 absolute tolerance."
          },
          {
            "check": "annotations",
            "status": "passed",
            "detail": "Registered subgroup fields derive from the preserved source snapshot."
          },
          {
            "check": "dependence",
            "status": "passed",
            "detail": "Independent sampling unit unknown; descriptive analysis only."
          },
          {
            "check": "recipe_pinned",
            "status": "passed",
            "detail": "Registered SDK recipe; implementation artifact and environment hashes retained."
          },
          {
            "check": "resource_estimate",
            "status": "passed",
            "detail": "Local CPU controls previously executed; all jobs remain subject to the campaign watchdog."
          }
        ],
        "limitations": [
          "Existing test outcomes are exposed; no independent validation is claimed.",
          "Independent experimental grouping is unavailable; subgroup summaries are descriptive.",
          "Prepared opaque IDs must not be joined to a newly prepared snapshot by ID.",
          "High wavelength is not universally biologically preferable; NDCG is a numerical ranking metric.",
          "Known methodological control, not a novel biological finding."
        ]
      },
      "local_recipes": [
        "sdk:train-mean-v1",
        "sdk:sequence-composition-v1"
      ]
    }
  ]
}
