{
  "artifacts": [
    {
      "format": "json",
      "id": "mrnabench-discrepancies-prepared",
      "role": "prepared",
      "semantic_sha256": "f4f9777b5119b739437c6a9d85e4e1b937268dafd1307efca5f072d38957738b",
      "sha256": "e5d6fb9e5fadec1f9e6235ca3ed2054ac05f476ccc8f5dc67ec500febeae8237",
      "uri": null
    },
    {
      "format": "json",
      "id": "mrnabench-discrepancies-mrnabench-composition-predictions",
      "role": "mrnabench-composition-predictions",
      "sha256": "a8f80cfea7ddd197ef9446e161f5a761741b3f778dfc1459275c3355432e15c5",
      "uri": null
    },
    {
      "format": "json",
      "id": "mrnabench-discrepancies-mrnabench-composition-report",
      "role": "mrnabench-composition-report",
      "sha256": "8b68dfe5725bf5d8c108265039e3787f4b3b55986eab8a430082c4d446a4f960",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-composition/report.json"
    },
    {
      "format": "json",
      "id": "mrnabench-discrepancies-mrnabench-composition-source",
      "role": "mrnabench-composition-source",
      "sha256": "ef486f7be79a6f5589afd7494b7acfa13ca698da50f0ffd0e50fd6eba1eb9934",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-composition/source.json"
    },
    {
      "format": "json",
      "id": "mrnabench-discrepancies-mrnabench-train-mean-predictions",
      "role": "mrnabench-train-mean-predictions",
      "sha256": "fb435ffda2e56792a9c781fba8dfda535b3d6508cf28f484162c5fd9065a87d8",
      "uri": null
    },
    {
      "format": "json",
      "id": "mrnabench-discrepancies-mrnabench-train-mean-report",
      "role": "mrnabench-train-mean-report",
      "sha256": "c036d9fa75b6ba4d5c65d31cd575ee2bfc357beec3ca7e6c80f07e3128a04e71",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-train-mean/report.json"
    },
    {
      "format": "json",
      "id": "mrnabench-discrepancies-mrnabench-train-mean-source",
      "role": "mrnabench-train-mean-source",
      "sha256": "b8bbc4f3f2ef54d92721015205bdd73a81e4cfcbc596985e8724be3416899d54",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-train-mean/source.json"
    },
    {
      "format": "py",
      "id": "mrnabench-discrepancies-recipe_code",
      "role": "recipe_code",
      "sha256": "b3407940ca4c8a1cdcf3a3bb3849c3e26d7f85272b2ee63087307378ffaa79dc",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/packages/rewirebench/src/rewirebench/adapters/sequence.py"
    },
    {
      "format": "json",
      "id": "mrnabench-discrepancies-table",
      "role": "table",
      "sha256": "d49ff1b4dacdfe0663c35bdad6a9ad91e79f5862fe96f36a627acb8407610bf3",
      "uri": null
    },
    {
      "format": "json",
      "id": "mrnabench-discrepancies-receipt",
      "role": "receipt",
      "sha256": "5edeb10e42be225f6b85481c95eb8985ab1724d3af353737fdafd29ccdfc4022",
      "uri": null
    },
    {
      "format": "py",
      "id": "mrnabench-discrepancies-preparation_code",
      "role": "preparation_code",
      "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
      "uri": null
    },
    {
      "format": "lock",
      "id": "mrnabench-discrepancies-environment",
      "role": "environment",
      "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
      "uri": null
    }
  ],
  "attempts": [
    {
      "error": null,
      "finished_at": "2026-09-23T23:19:57.075139Z",
      "id": "attempt-4c2a8a6a7510aeac14ad60cb",
      "operation": {
        "expected_observation": "All available evidence matches its declared byte hash.",
        "field": null,
        "id": "verify-evidence",
        "kind": "verify",
        "methods": [],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "a93df2292a2c9acdbbaab9d8363fa6a2998771e2a12e41934e1e05076c084581",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "verify",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "28244c39debc78e33a6e5fd85f1323a80f235916f3a51e42e257dd007d160fa9",
        "numerical": {
          "checks": [
            {
              "artifact_id": "mrnabench-discrepancies-prepared",
              "status": "passed"
            },
            {
              "artifact_id": "mrnabench-discrepancies-mrnabench-composition-predictions",
              "status": "passed"
            },
            {
              "artifact_id": "mrnabench-discrepancies-mrnabench-composition-report",
              "status": "passed"
            },
            {
              "artifact_id": "mrnabench-discrepancies-mrnabench-composition-source",
              "status": "passed"
            },
            {
              "artifact_id": "mrnabench-discrepancies-mrnabench-train-mean-predictions",
              "status": "passed"
            },
            {
              "artifact_id": "mrnabench-discrepancies-mrnabench-train-mean-report",
              "status": "passed"
            },
            {
              "artifact_id": "mrnabench-discrepancies-mrnabench-train-mean-source",
              "status": "passed"
            },
            {
              "artifact_id": "mrnabench-discrepancies-recipe_code",
              "status": "passed"
            },
            {
              "artifact_id": "mrnabench-discrepancies-table",
              "status": "passed"
            },
            {
              "artifact_id": "mrnabench-discrepancies-receipt",
              "status": "passed"
            },
            {
              "artifact_id": "mrnabench-discrepancies-preparation_code",
              "status": "passed"
            },
            {
              "artifact_id": "mrnabench-discrepancies-environment",
              "status": "passed"
            }
          ],
          "identifiers_unique": true,
          "methods": [
            "mrnabench-composition",
            "mrnabench-train-mean"
          ],
          "rows": 15003
        },
        "operation_id": "verify-evidence",
        "table_sha256": "d49ff1b4dacdfe0663c35bdad6a9ad91e79f5862fe96f36a627acb8407610bf3"
      },
      "receipt_sha256": "bb668f35d0dad66413e6d16be7ed923a731da3e4cb1b5dbbf5fe8b20dd9561a6",
      "started_at": "2026-09-23T23:19:56.155719Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:19:57.883873Z",
      "id": "attempt-d5f741e912b1becf73ce1c08",
      "operation": {
        "expected_observation": "Recorded metrics replay within the stated tolerance.",
        "field": null,
        "id": "replay-metrics",
        "kind": "replay",
        "methods": [],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "a93df2292a2c9acdbbaab9d8363fa6a2998771e2a12e41934e1e05076c084581",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "replay",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "28244c39debc78e33a6e5fd85f1323a80f235916f3a51e42e257dd007d160fa9",
        "numerical": {
          "checks": [
            {
              "actual": 1.914439715839851,
              "expected": 1.914439715839851,
              "method": "mrnabench-composition",
              "metric": "mse",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.436786917017099,
              "expected": 0.4367869170170987,
              "method": "mrnabench-composition",
              "metric": "pearson",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.49477537290902324,
              "expected": 0.49477537290902324,
              "method": "mrnabench-composition",
              "metric": "spearman",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 2.3655294722554605,
              "expected": 2.3655294722554605,
              "method": "mrnabench-train-mean",
              "metric": "mse",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": null,
              "expected": null,
              "method": "mrnabench-train-mean",
              "metric": "pearson",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": null,
              "expected": null,
              "method": "mrnabench-train-mean",
              "metric": "spearman",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            }
          ],
          "coverage": {
            "mrnabench-composition": {
              "denominator": 15003,
              "missing": 0,
              "scored": 15003
            },
            "mrnabench-train-mean": {
              "denominator": 15003,
              "missing": 0,
              "scored": 15003
            }
          },
          "metrics": {
            "mrnabench-composition": {
              "mse": 1.914439715839851,
              "ndcg": 0.9837668716857858,
              "pearson": 0.436786917017099,
              "spearman": 0.49477537290902324
            },
            "mrnabench-train-mean": {
              "mse": 2.3655294722554605,
              "ndcg": 0.9687108424175718,
              "pearson": null,
              "spearman": null
            }
          }
        },
        "operation_id": "replay-metrics",
        "table_sha256": "d49ff1b4dacdfe0663c35bdad6a9ad91e79f5862fe96f36a627acb8407610bf3"
      },
      "receipt_sha256": "0ecb2751dcff85b0a79c1b185af36a86e200d5dfc0e205e7f5cc78d7756caf4f",
      "started_at": "2026-09-23T23:19:57.076029Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:20:01.939317Z",
      "id": "attempt-d6dee0a62da532c2c1bfd3dc",
      "operation": {
        "expected_observation": "The SDK trains only on training labels and reproduces the constant control.",
        "field": null,
        "id": "local-train-mean",
        "kind": "local_recipe",
        "methods": [],
        "metric": null,
        "recipe": "sdk:train-mean-v1"
      },
      "plan_sha256": "a93df2292a2c9acdbbaab9d8363fa6a2998771e2a12e41934e1e05076c084581",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "local_recipe",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "28244c39debc78e33a6e5fd85f1323a80f235916f3a51e42e257dd007d160fa9",
        "numerical": {
          "coverage": {
            "denominator": 15003,
            "scored": 15003,
            "unscored": 0
          },
          "fitting": "train_only",
          "metrics": {
            "mse": 2.3655294722554605,
            "pearson": null,
            "spearman": null
          },
          "prepared_sha256": "f4f9777b5119b739437c6a9d85e4e1b937268dafd1307efca5f072d38957738b",
          "protocol_id": "mrnabench-sample-mrl-v1",
          "recipe": "sdk:train-mean-v1",
          "report_sha256": "3add1ff9939b9c2248e171adbde5ab1a4e2b3005486b45dda033367773963fde",
          "runtime": {
            "adapter_sha256": "b3407940ca4c8a1cdcf3a3bb3849c3e26d7f85272b2ee63087307378ffaa79dc",
            "dependencies": {
              "numpy": "1.26.4",
              "scikit-learn": "1.9.1",
              "scipy": "1.17.1"
            },
            "platform": "Darwin",
            "python": "3.11.13",
            "runner_code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596"
          },
          "scope": "full"
        },
        "operation_id": "local-train-mean",
        "table_sha256": "d49ff1b4dacdfe0663c35bdad6a9ad91e79f5862fe96f36a627acb8407610bf3"
      },
      "receipt_sha256": "5c9f4edf395b1e9b1208dcae9e4bf4e661eac7ba6b08a9b096e264811fa78202",
      "started_at": "2026-09-23T23:19:57.884664Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:20:02.739946Z",
      "id": "attempt-415c28fe703ad1225e7e7b13",
      "operation": {
        "expected_observation": "Unequal scored or missing counts support a possible population mismatch. Complete coverage for both methods rules out this explanation. Equal incomplete counts do not establish identical scored rows.",
        "field": null,
        "id": "agent-1-1-whole-population-coverage",
        "kind": "coverage",
        "methods": [
          "mrnabench-composition",
          "mrnabench-train-mean"
        ],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "a93df2292a2c9acdbbaab9d8363fa6a2998771e2a12e41934e1e05076c084581",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "coverage",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "28244c39debc78e33a6e5fd85f1323a80f235916f3a51e42e257dd007d160fa9",
        "numerical": {
          "common_all_methods": 15003,
          "coverage": {
            "mrnabench-composition": {
              "denominator": 15003,
              "missing": 0,
              "scored": 15003
            },
            "mrnabench-train-mean": {
              "denominator": 15003,
              "missing": 0,
              "scored": 15003
            }
          },
          "original_n": 15003
        },
        "operation_id": "agent-1-1-whole-population-coverage",
        "table_sha256": "d49ff1b4dacdfe0663c35bdad6a9ad91e79f5862fe96f36a627acb8407610bf3"
      },
      "receipt_sha256": "c9a669b64a2f8643561de9dc47ae021971f140aaec2f2e00d6a525dac7001b34",
      "started_at": "2026-09-23T23:20:01.940369Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:20:03.549589Z",
      "id": "attempt-7ae9cab8c67a3217201e72e6",
      "operation": {
        "expected_observation": "With composition as candidate and training-mean as reference, negative candidate-minus-reference MSE supports improvement on common scored IDs; zero or positive differences contradict improvement there. Persistence with complete coverage weakens the population-mismatch explanation.",
        "field": null,
        "id": "agent-1-2-common-row-mse",
        "kind": "paired",
        "methods": [
          "mrnabench-composition",
          "mrnabench-train-mean"
        ],
        "metric": "mse",
        "recipe": null
      },
      "plan_sha256": "a93df2292a2c9acdbbaab9d8363fa6a2998771e2a12e41934e1e05076c084581",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "paired",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "28244c39debc78e33a6e5fd85f1323a80f235916f3a51e42e257dd007d160fa9",
        "numerical": {
          "pairs": [
            {
              "candidate": "mrnabench-train-mean",
              "candidate_metrics": {
                "mse": 2.3655294722554605,
                "ndcg": 0.9687108424175718,
                "pearson": null,
                "spearman": null
              },
              "candidate_minus_reference": {
                "mse": 0.45108975641560956
              },
              "common_n": 15003,
              "original_n": 15003,
              "reference": "mrnabench-composition",
              "reference_metrics": {
                "mse": 1.914439715839851,
                "ndcg": 0.9837668716857858,
                "pearson": 0.436786917017099,
                "spearman": 0.49477537290902324
              }
            }
          ]
        },
        "operation_id": "agent-1-2-common-row-mse",
        "table_sha256": "d49ff1b4dacdfe0663c35bdad6a9ad91e79f5862fe96f36a627acb8407610bf3"
      },
      "receipt_sha256": "01248d1a792693335f207dd843fd4bb6569d77e9149a80ac1cbd022db12a25c2",
      "started_at": "2026-09-23T23:20:02.740577Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:20:04.357446Z",
      "id": "attempt-f0efeff6e46e021b0e80c9d8",
      "operation": {
        "expected_observation": "Greater MSE reductions in duplicate-labelled categories, with little or no reduction elsewhere, support concentration. Comparable reductions across duplicate categories weaken this explanation. Category counts, outcome means and prediction coverage provide context for each comparison.",
        "field": "duplicate_sequence",
        "id": "agent-2-1-duplicate-category-performance",
        "kind": "subgroups",
        "methods": [
          "mrnabench-composition",
          "mrnabench-train-mean"
        ],
        "metric": "mse",
        "recipe": null
      },
      "plan_sha256": "a93df2292a2c9acdbbaab9d8363fa6a2998771e2a12e41934e1e05076c084581",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "subgroups",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "28244c39debc78e33a6e5fd85f1323a80f235916f3a51e42e257dd007d160fa9",
        "numerical": {
          "binning": {
            "edges": [],
            "rule": "predeclared annotation categories"
          },
          "field": "duplicate_sequence",
          "subgroups": [
            {
              "label": "no",
              "methods": {
                "mrnabench-composition": {
                  "metrics": {
                    "mse": 1.914439715839851,
                    "ndcg": 0.9837668716857858,
                    "pearson": 0.436786917017099,
                    "spearman": 0.49477537290902324
                  },
                  "scored": 15003
                },
                "mrnabench-train-mean": {
                  "metrics": {
                    "mse": 2.3655294722554605,
                    "ndcg": 0.9687108424175718,
                    "pearson": null,
                    "spearman": null
                  },
                  "scored": 15003
                }
              },
              "n": 15003,
              "outcome_mean": 5.71319217945288
            }
          ]
        },
        "operation_id": "agent-2-1-duplicate-category-performance",
        "table_sha256": "d49ff1b4dacdfe0663c35bdad6a9ad91e79f5862fe96f36a627acb8407610bf3"
      },
      "receipt_sha256": "0c5740c791fe53d49f296372ad0f84e3f7e6d6cf0c926b12bd4ed13567254326",
      "started_at": "2026-09-23T23:20:03.550328Z",
      "status": "completed"
    }
  ],
  "catalogue_release_id": "2026-09-20-370b30415b09",
  "claim_level": "exploratory",
  "created_at": "2026-09-23T23:00:33.952383Z",
  "execution": {
    "campaign_id": "campaign-fd29c4413eaf4e24",
    "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
    "codex_calls": 2,
    "reasoning": [
      {
        "cli_version": "codex-cli 0.155.0-alpha.9.2",
        "model": "gpt-6-astra",
        "role": "planner",
        "usage": {
          "cache_write_input_tokens": 0,
          "cached_input_tokens": 0,
          "input_tokens": 12252,
          "output_tokens": 8861,
          "reasoning_output_tokens": 8286
        }
      },
      {
        "cli_version": "codex-cli 0.155.0-alpha.9.2",
        "model": "gpt-6-astra",
        "role": "critic",
        "usage": {
          "cache_write_input_tokens": 0,
          "cached_input_tokens": 0,
          "input_tokens": 17002,
          "output_tokens": 3500,
          "reasoning_output_tokens": 3106
        }
      }
    ]
  },
  "findings": [
    "Across 15,003 common rows, mrnabench-composition has mse 1.9144397158 versus 2.3655294723 for mrnabench-train-mean, a reduction of 0.4510897564. Both have complete coverage, contradicting unequal prediction coverage as an explanation.",
    "The paired receipt reverses the planned roles: candidate=mrnabench-train-mean and reference=mrnabench-composition. Its positive candidate-minus-reference mse therefore favors composition.",
    "All 12 artifact checks and six recorded metric replay checks passed. sdk:train-mean-v1 reports fitting=train_only and reproduces the control mse. These results support internal consistency.",
    "All 15,003 rows are annotated duplicate_sequence=no. Duplicate-labelled rows cannot account for the observed gain within this table, but performance differences between duplicate categories cannot be estimated."
  ],
  "id": "mrnabench-discrepancies-campaign-fd29c4413eaf4e24",
  "limitations": [
    "Existing test outcomes are exposed; no independent validation is claimed.",
    "Independent experimental grouping is unavailable; subgroup summaries are descriptive.",
    "Prepared opaque IDs must not be joined to a newly prepared snapshot by ID.",
    "Random split does not establish homology separation.",
    "Source data reuse licence is unreported; no raw data redistribution.",
    "Exploratory analysis of exposed outcomes; no independent validation or novelty claim.",
    "AI-assisted interpretation requires human scientific review before publication.",
    "All intervals are descriptive and unadjusted; unsuccessful tests remain in the report.",
    "No length_band or gc_band subgroup receipts were supplied. Aggregate improvement is established; its concentration within the evaluated population remains unresolved, and improvement on every row is not established.",
    "Reproducing the training-mean mse does not establish baseline adequacy or demonstrate that mrnabench-composition was fitted using training data only.",
    "The duplicate annotation does not establish absence of train/test leakage or homology overlap. Hash and replay checks do not establish protocol validity.",
    "The control's pearson and spearman are null, not zero; numerical correlation improvements over it cannot be calculated.",
    "These are previously exposed data with undocumented sampling independence. Findings remain descriptive; bootstrap intervals, statistical significance, mechanisms and independent validation are unsupported. No follow-up rounds remain."
  ],
  "manifest_id": "mrnabench-discrepancies",
  "outcome": "inconclusive",
  "plan_sha256": "746d3251cfd3ff5e530338fcde057903b72e43cdef411ca82b46355667558147",
  "question": "Where does the composition predictor improve over the training-mean control?",
  "review": {
    "method": "ai_assisted",
    "status": "pending"
  },
  "schema_version": "1.0",
  "specs": [
    {
      "budget": {
        "campaign_seconds": 3600,
        "codex_calls": 8,
        "codex_seconds": 900,
        "experiment_seconds": 300,
        "followup_rounds": 0,
        "memory_bytes": 2147483648,
        "workspace_bytes": 1073741824
      },
      "catalogue_release_id": "2026-09-20-370b30415b09",
      "created_at": "2026-09-23T23:19:56.148786Z",
      "evidence": [
        {
          "format": "json",
          "id": "mrnabench-discrepancies-prepared",
          "role": "prepared",
          "semantic_sha256": "f4f9777b5119b739437c6a9d85e4e1b937268dafd1307efca5f072d38957738b",
          "sha256": "e5d6fb9e5fadec1f9e6235ca3ed2054ac05f476ccc8f5dc67ec500febeae8237",
          "uri": null
        },
        {
          "format": "json",
          "id": "mrnabench-discrepancies-mrnabench-composition-predictions",
          "role": "mrnabench-composition-predictions",
          "sha256": "a8f80cfea7ddd197ef9446e161f5a761741b3f778dfc1459275c3355432e15c5",
          "uri": null
        },
        {
          "format": "json",
          "id": "mrnabench-discrepancies-mrnabench-composition-report",
          "role": "mrnabench-composition-report",
          "sha256": "8b68dfe5725bf5d8c108265039e3787f4b3b55986eab8a430082c4d446a4f960",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-composition/report.json"
        },
        {
          "format": "json",
          "id": "mrnabench-discrepancies-mrnabench-composition-source",
          "role": "mrnabench-composition-source",
          "sha256": "ef486f7be79a6f5589afd7494b7acfa13ca698da50f0ffd0e50fd6eba1eb9934",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-composition/source.json"
        },
        {
          "format": "json",
          "id": "mrnabench-discrepancies-mrnabench-train-mean-predictions",
          "role": "mrnabench-train-mean-predictions",
          "sha256": "fb435ffda2e56792a9c781fba8dfda535b3d6508cf28f484162c5fd9065a87d8",
          "uri": null
        },
        {
          "format": "json",
          "id": "mrnabench-discrepancies-mrnabench-train-mean-report",
          "role": "mrnabench-train-mean-report",
          "sha256": "c036d9fa75b6ba4d5c65d31cd575ee2bfc357beec3ca7e6c80f07e3128a04e71",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-train-mean/report.json"
        },
        {
          "format": "json",
          "id": "mrnabench-discrepancies-mrnabench-train-mean-source",
          "role": "mrnabench-train-mean-source",
          "sha256": "b8bbc4f3f2ef54d92721015205bdd73a81e4cfcbc596985e8724be3416899d54",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/mrnabench-train-mean/source.json"
        },
        {
          "format": "py",
          "id": "mrnabench-discrepancies-recipe_code",
          "role": "recipe_code",
          "sha256": "b3407940ca4c8a1cdcf3a3bb3849c3e26d7f85272b2ee63087307378ffaa79dc",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/packages/rewirebench/src/rewirebench/adapters/sequence.py"
        },
        {
          "format": "json",
          "id": "mrnabench-discrepancies-table",
          "role": "table",
          "sha256": "d49ff1b4dacdfe0663c35bdad6a9ad91e79f5862fe96f36a627acb8407610bf3",
          "uri": null
        },
        {
          "format": "json",
          "id": "mrnabench-discrepancies-receipt",
          "role": "receipt",
          "sha256": "5edeb10e42be225f6b85481c95eb8985ab1724d3af353737fdafd29ccdfc4022",
          "uri": null
        },
        {
          "format": "py",
          "id": "mrnabench-discrepancies-preparation_code",
          "role": "preparation_code",
          "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
          "uri": null
        },
        {
          "format": "lock",
          "id": "mrnabench-discrepancies-environment",
          "role": "environment",
          "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
          "uri": null
        }
      ],
      "exposure": {
        "independent_validation": false,
        "previously_exposed": true,
        "usage": "exploration"
      },
      "id": "mrnabench-discrepancies-round-0",
      "manifest_id": "mrnabench-discrepancies",
      "manifest_sha256": "28244c39debc78e33a6e5fd85f1323a80f235916f3a51e42e257dd007d160fa9",
      "permitted_actions": [
        "verify",
        "replay",
        "coverage",
        "paired",
        "subgroups",
        "bootstrap",
        "sensitivity",
        "local_recipe"
      ],
      "plan": {
        "hypotheses": [
          {
            "explanation": "The discrepancy may arise from stale evidence, metric calculation or an inadequate trivial control.",
            "id": "evidence-and-method",
            "tests": [
              {
                "expected_observation": "All available evidence matches its declared byte hash.",
                "field": null,
                "id": "verify-evidence",
                "kind": "verify",
                "methods": [],
                "metric": null,
                "recipe": null
              },
              {
                "expected_observation": "Recorded metrics replay within the stated tolerance.",
                "field": null,
                "id": "replay-metrics",
                "kind": "replay",
                "methods": [],
                "metric": null,
                "recipe": null
              },
              {
                "expected_observation": "The SDK trains only on training labels and reproduces the constant control.",
                "field": null,
                "id": "local-train-mean",
                "kind": "local_recipe",
                "methods": [],
                "metric": null,
                "recipe": "sdk:train-mean-v1"
              }
            ]
          },
          {
            "explanation": "Unequal prediction coverage could explain part of the apparent composition advantage.",
            "id": "agent-1-evaluation-population",
            "tests": [
              {
                "expected_observation": "Unequal scored or missing counts support a possible population mismatch. Complete coverage for both methods rules out this explanation. Equal incomplete counts do not establish identical scored rows.",
                "field": null,
                "id": "agent-1-1-whole-population-coverage",
                "kind": "coverage",
                "methods": [
                  "mrnabench-composition",
                  "mrnabench-train-mean"
                ],
                "metric": null,
                "recipe": null
              },
              {
                "expected_observation": "With composition as candidate and training-mean as reference, negative candidate-minus-reference MSE supports improvement on common scored IDs; zero or positive differences contradict improvement there. Persistence with complete coverage weakens the population-mismatch explanation.",
                "field": null,
                "id": "agent-1-2-common-row-mse",
                "kind": "paired",
                "methods": [
                  "mrnabench-composition",
                  "mrnabench-train-mean"
                ],
                "metric": "mse",
                "recipe": null
              }
            ]
          },
          {
            "explanation": "Any pooled improvement may be concentrated in duplicate-labelled sequences. This annotation alone cannot establish train/test leakage.",
            "id": "agent-2-duplicate-concentration",
            "tests": [
              {
                "expected_observation": "Greater MSE reductions in duplicate-labelled categories, with little or no reduction elsewhere, support concentration. Comparable reductions across duplicate categories weaken this explanation. Category counts, outcome means and prediction coverage provide context for each comparison.",
                "field": "duplicate_sequence",
                "id": "agent-2-1-duplicate-category-performance",
                "kind": "subgroups",
                "methods": [
                  "mrnabench-composition",
                  "mrnabench-train-mean"
                ],
                "metric": "mse",
                "recipe": null
              }
            ]
          }
        ],
        "multiple_testing": "descriptive_only",
        "requested_tools": [],
        "stopping_rule": "Interpret additional tests only after the supervisor checks pass. Treat manifest metrics as targets until numerical receipts establish results. Stop after these three additional tests and retain contradicted explanations and unavailable results. Compare subgroup MSE differences only when both methods score every row in that category; otherwise report localization as unresolved. Retain predeclared categories and use fixed annotation-only quartiles for numeric annotations. Unknown sampling independence precludes bootstrap. Results remain exploratory on exposed data, without significance, mechanism or independent-validation claims. Request follow-up only when an unused registered test would materially resolve a specific remaining uncertainty."
      },
      "question": "Where does the composition predictor improve over the training-mean control?",
      "round": 0,
      "schema_version": "1.0"
    }
  ],
  "status": "completed",
  "title": "mRNABench designed: composition effects"
}
