{
  "artifacts": [
    {
      "format": "tsv",
      "id": "mfass-v2-discrepancies-cohort",
      "role": "cohort",
      "sha256": "389702ff4c647d7ce10a90092a6fa811ae777d15997baf39ce9aae0346247bd0",
      "uri": null
    },
    {
      "format": "txt",
      "id": "mfass-v2-discrepancies-outcomes",
      "role": "outcomes",
      "sha256": "a637ca0e307e66ff48811ec7efa22b9ce453bc7883b04f0cacb867f7283132d8",
      "uri": "https://raw.githubusercontent.com/KosuriLab/MFASS/master/processed_data/snv/snv_data_clean.txt"
    },
    {
      "format": "txt",
      "id": "mfass-v2-discrepancies-annotations",
      "role": "annotations",
      "sha256": "71a857fe647c4e68acbb41ca61e959c47e1176de89b1442bd6ca1772aa60d5a1",
      "uri": "https://raw.githubusercontent.com/KosuriLab/MFASS/master/processed_data/snv/snv_func_annot.txt"
    },
    {
      "format": "tsv",
      "id": "mfass-v2-discrepancies-split",
      "role": "split",
      "sha256": "999ebcb7e63a5c5eaa8780fa468e59ac1f934260ad50102814174c396317f052",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/splits/split-v2.tsv"
    },
    {
      "format": "npy",
      "id": "mfass-v2-discrepancies-baseline-kmer-position-v2-scores",
      "role": "baseline-kmer-position-v2-scores",
      "sha256": "ed0ebb3d74183deb1fb475d0e706c0b1211fd6ab9d20449c2c194e57f912d862",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/baseline-kmer-position-v2.scores.npy"
    },
    {
      "format": "tsv",
      "id": "mfass-v2-discrepancies-baseline-kmer-position-v2-predictions",
      "role": "baseline-kmer-position-v2-predictions",
      "sha256": "2f3117c225a8da9ea737abfa2d0f7e1dec97696ae4bacd263864bec9d7f923eb",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/baseline-kmer-position-v2.predictions.tsv"
    },
    {
      "format": "json",
      "id": "mfass-v2-discrepancies-baseline-kmer-position-v2-report",
      "role": "baseline-kmer-position-v2-report",
      "sha256": "9a0b78674cc714177fec6e8c487588d6186d4fe93d6d7dbcb48bed6858885e15",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/baseline-kmer-position-v2.json"
    },
    {
      "format": "npy",
      "id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-scores",
      "role": "dnabert2-117m-frozen-pair-logreg-scores",
      "sha256": "e6b019babf0b3fb9ec05d4261a9383c0c896183a76c3b31634564c7a63d01105",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.scores.npy"
    },
    {
      "format": "tsv",
      "id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-predictions",
      "role": "dnabert2-117m-frozen-pair-logreg-predictions",
      "sha256": "3abded2932366e6e2c8d6e0da5836c4240239372758b3c68cd5f19635b8cee11",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.predictions.tsv"
    },
    {
      "format": "json",
      "id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-report",
      "role": "dnabert2-117m-frozen-pair-logreg-report",
      "sha256": "60b28541853de349f878d6d1ccbcbe7dfd21b69db976e9c01f60e88bdba0f2a1",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.json"
    },
    {
      "format": "tsv",
      "id": "mfass-v2-discrepancies-spliceai-1.3.1-predictions",
      "role": "spliceai-1.3.1-predictions",
      "sha256": "b39df773067e4ecd862980f177da69d6117c75c93c754f4cbdd7a77bd419851f",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/spliceai-1.3.1.predictions.tsv"
    },
    {
      "format": "json",
      "id": "mfass-v2-discrepancies-spliceai-1.3.1-report",
      "role": "spliceai-1.3.1-report",
      "sha256": "6d5c59eb0fd60d95064d97e331c9cdb12b2a34db7dc509ff73dfb9502ddf02dd",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/spliceai-1.3.1.json"
    },
    {
      "format": "tsv",
      "id": "mfass-v2-discrepancies-pangolin-maskFalse-predictions",
      "role": "pangolin-maskFalse-predictions",
      "sha256": "faa7d4cb3728111f1c5a8e8be6cd2c5958c4cced7c7a366ec96171d03b855ad6",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/pangolin-maskFalse.predictions.tsv"
    },
    {
      "format": "json",
      "id": "mfass-v2-discrepancies-pangolin-maskFalse-report",
      "role": "pangolin-maskFalse-report",
      "sha256": "bb0bb6732808699e54938233df1835dfc1f775f33ba7d6acd916e53d588a3c44",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/pangolin-maskFalse.json"
    },
    {
      "format": "json",
      "id": "mfass-v2-discrepancies-table",
      "role": "table",
      "sha256": "c57f9cfe91d02ab54f6bc20934f5b8f678b37c3b13d8d20aa805fa7d73e62c4f",
      "uri": null
    },
    {
      "format": "json",
      "id": "mfass-v2-discrepancies-receipt",
      "role": "receipt",
      "sha256": "a9fa54bcf99457f9b23424ad656a952eb262b376845327e267dc035cf85a012e",
      "uri": null
    },
    {
      "format": "py",
      "id": "mfass-v2-discrepancies-preparation_code",
      "role": "preparation_code",
      "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
      "uri": null
    },
    {
      "format": "lock",
      "id": "mfass-v2-discrepancies-environment",
      "role": "environment",
      "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
      "uri": null
    }
  ],
  "attempts": [
    {
      "error": null,
      "finished_at": "2026-09-23T23:12:53.847945Z",
      "id": "attempt-d75ef3151ae5341a77d1d07b",
      "operation": {
        "expected_observation": "All available evidence matches its declared byte hash.",
        "field": null,
        "id": "verify-evidence",
        "kind": "verify",
        "methods": [],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "dc65f50a8617ad1c2d3cf4e75e01dcb67574076abbf8094a7350ed9b53842e42",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "verify",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "4fca3c60f42b6f2d4259b0aa2c0c339d05d32b5db9f191159fa1b41ac9a97e64",
        "numerical": {
          "checks": [
            {
              "artifact_id": "mfass-v2-discrepancies-cohort",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-outcomes",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-annotations",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-split",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-baseline-kmer-position-v2-scores",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-baseline-kmer-position-v2-predictions",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-baseline-kmer-position-v2-report",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-scores",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-predictions",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-report",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-spliceai-1.3.1-predictions",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-spliceai-1.3.1-report",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-pangolin-maskFalse-predictions",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-pangolin-maskFalse-report",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-table",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-receipt",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-preparation_code",
              "status": "passed"
            },
            {
              "artifact_id": "mfass-v2-discrepancies-environment",
              "status": "passed"
            }
          ],
          "identifiers_unique": true,
          "methods": [
            "baseline-kmer-position-v2",
            "dnabert2-117m-frozen-pair-logreg",
            "pangolin-maskFalse",
            "spliceai-1.3.1"
          ],
          "rows": 8324
        },
        "operation_id": "verify-evidence",
        "table_sha256": "c57f9cfe91d02ab54f6bc20934f5b8f678b37c3b13d8d20aa805fa7d73e62c4f"
      },
      "receipt_sha256": "9826b8bfed2a384611374436266727332fe6d2c07606ae3898ddcc68dbff5e0d",
      "started_at": "2026-09-23T23:12:52.946510Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:12:54.645479Z",
      "id": "attempt-6aa2726c5e69e7c10181de53",
      "operation": {
        "expected_observation": "Recorded metrics replay within the stated tolerance.",
        "field": null,
        "id": "replay-metrics",
        "kind": "replay",
        "methods": [],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "dc65f50a8617ad1c2d3cf4e75e01dcb67574076abbf8094a7350ed9b53842e42",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "replay",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "4fca3c60f42b6f2d4259b0aa2c0c339d05d32b5db9f191159fa1b41ac9a97e64",
        "numerical": {
          "checks": [
            {
              "actual": 0.7779498064677238,
              "expected": 0.7779498064677238,
              "method": "baseline-kmer-position-v2",
              "metric": "auroc",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.2864167459589237,
              "expected": 0.2864167459589237,
              "method": "baseline-kmer-position-v2",
              "metric": "average_precision_sklearn",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.61,
              "expected": 0.61,
              "method": "baseline-kmer-position-v2",
              "metric": "precision_at_capacity",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.19365079365079366,
              "expected": 0.19365079365079366,
              "method": "baseline-kmer-position-v2",
              "metric": "recall_at_capacity",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.5500324040216661,
              "expected": 0.5500324040216661,
              "method": "dnabert2-117m-frozen-pair-logreg",
              "metric": "auroc",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.04508654312652131,
              "expected": 0.04508654312652131,
              "method": "dnabert2-117m-frozen-pair-logreg",
              "metric": "average_precision_sklearn",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.03,
              "expected": 0.03,
              "method": "dnabert2-117m-frozen-pair-logreg",
              "metric": "precision_at_capacity",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.009523809523809525,
              "expected": 0.009523809523809525,
              "method": "dnabert2-117m-frozen-pair-logreg",
              "metric": "recall_at_capacity",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.8756851300560864,
              "expected": 0.8756851300560864,
              "method": "pangolin-maskFalse",
              "metric": "auroc",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.3887617543064248,
              "expected": 0.3887617543064248,
              "method": "pangolin-maskFalse",
              "metric": "average_precision_sklearn",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.65,
              "expected": 0.65,
              "method": "pangolin-maskFalse",
              "metric": "precision_at_capacity",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.2070063694267516,
              "expected": 0.2070063694267516,
              "method": "pangolin-maskFalse",
              "metric": "recall_at_capacity",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.8055241740253153,
              "expected": 0.8055241740253153,
              "method": "spliceai-1.3.1",
              "metric": "auroc",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.2986855472760137,
              "expected": 0.2986855472760137,
              "method": "spliceai-1.3.1",
              "metric": "average_precision_sklearn",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.64,
              "expected": 0.64,
              "method": "spliceai-1.3.1",
              "metric": "precision_at_capacity",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.2077922077922078,
              "expected": 0.2077922077922078,
              "method": "spliceai-1.3.1",
              "metric": "recall_at_capacity",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            }
          ],
          "coverage": {
            "baseline-kmer-position-v2": {
              "denominator": 8324,
              "missing": 0,
              "scored": 8324
            },
            "dnabert2-117m-frozen-pair-logreg": {
              "denominator": 8324,
              "missing": 0,
              "scored": 8324
            },
            "pangolin-maskFalse": {
              "denominator": 8324,
              "missing": 23,
              "scored": 8301
            },
            "spliceai-1.3.1": {
              "denominator": 8324,
              "missing": 130,
              "scored": 8194
            }
          },
          "metrics": {
            "baseline-kmer-position-v2": {
              "auroc": 0.7779498064677238,
              "average_precision_sklearn": 0.2864167459589237,
              "precision_at_capacity": 0.61,
              "recall_at_capacity": 0.19365079365079366
            },
            "dnabert2-117m-frozen-pair-logreg": {
              "auroc": 0.5500324040216661,
              "average_precision_sklearn": 0.04508654312652131,
              "precision_at_capacity": 0.03,
              "recall_at_capacity": 0.009523809523809525
            },
            "pangolin-maskFalse": {
              "auroc": 0.8756851300560864,
              "average_precision_sklearn": 0.3887617543064248,
              "precision_at_capacity": 0.65,
              "recall_at_capacity": 0.2070063694267516
            },
            "spliceai-1.3.1": {
              "auroc": 0.8055241740253153,
              "average_precision_sklearn": 0.2986855472760137,
              "precision_at_capacity": 0.64,
              "recall_at_capacity": 0.2077922077922078
            }
          }
        },
        "operation_id": "replay-metrics",
        "table_sha256": "c57f9cfe91d02ab54f6bc20934f5b8f678b37c3b13d8d20aa805fa7d73e62c4f"
      },
      "receipt_sha256": "bcb4a3f564c8e1f999dbd7f6291999f24786c0af041c4ea56bd09fcd515a20b6",
      "started_at": "2026-09-23T23:12:53.848886Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:12:55.444588Z",
      "id": "attempt-9594cf959716cd0d160e7d5b",
      "operation": {
        "expected_observation": "Different missing-prediction counts would support a denominator concern requiring the common-row comparison. Complete coverage for every method would contradict missingness as an explanation. Coverage alone cannot establish selection bias.",
        "field": null,
        "id": "agent-1-1-whole-population-coverage",
        "kind": "coverage",
        "methods": [
          "baseline-kmer-position-v2",
          "dnabert2-117m-frozen-pair-logreg",
          "pangolin-maskFalse",
          "spliceai-1.3.1"
        ],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "dc65f50a8617ad1c2d3cf4e75e01dcb67574076abbf8094a7350ed9b53842e42",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "coverage",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "4fca3c60f42b6f2d4259b0aa2c0c339d05d32b5db9f191159fa1b41ac9a97e64",
        "numerical": {
          "common_all_methods": 8194,
          "coverage": {
            "baseline-kmer-position-v2": {
              "denominator": 8324,
              "missing": 0,
              "scored": 8324
            },
            "dnabert2-117m-frozen-pair-logreg": {
              "denominator": 8324,
              "missing": 0,
              "scored": 8324
            },
            "pangolin-maskFalse": {
              "denominator": 8324,
              "missing": 23,
              "scored": 8301
            },
            "spliceai-1.3.1": {
              "denominator": 8324,
              "missing": 130,
              "scored": 8194
            }
          },
          "original_n": 8324
        },
        "operation_id": "agent-1-1-whole-population-coverage",
        "table_sha256": "c57f9cfe91d02ab54f6bc20934f5b8f678b37c3b13d8d20aa805fa7d73e62c4f"
      },
      "receipt_sha256": "a7b754abbf55d9f201eecb45ff3cbdd2254c786eab44ad542ce5f0bf0f547a9c",
      "started_at": "2026-09-23T23:12:54.646480Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:12:56.334163Z",
      "id": "attempt-a508a53d973c18b57fef6752",
      "operation": {
        "expected_observation": "Few unique scores and performance resembling the registered constant-ranking control would support score collapse or weak ranking information. Many unique scores would contradict collapse, though not weak information. Better reversed-score performance would flag a semantics question without establishing an error or authorizing a direction change.",
        "field": null,
        "id": "agent-1-2-score-information-diagnostics",
        "kind": "sensitivity",
        "methods": [
          "baseline-kmer-position-v2",
          "dnabert2-117m-frozen-pair-logreg",
          "pangolin-maskFalse",
          "spliceai-1.3.1"
        ],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "dc65f50a8617ad1c2d3cf4e75e01dcb67574076abbf8094a7350ed9b53842e42",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "sensitivity",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests.",
          "Sign reversal is diagnostic and never changes the declared score direction."
        ],
        "manifest_sha256": "4fca3c60f42b6f2d4259b0aa2c0c339d05d32b5db9f191159fa1b41ac9a97e64",
        "numerical": {
          "methods": {
            "baseline-kmer-position-v2": {
              "constant_ranking_control": {
                "auroc": 0.5,
                "average_precision_sklearn": 0.037842383469485825,
                "precision_at_capacity": 0.04,
                "recall_at_capacity": 0.012698412698412698
              },
              "original": {
                "auroc": 0.7779498064677238,
                "average_precision_sklearn": 0.2864167459589237,
                "precision_at_capacity": 0.61,
                "recall_at_capacity": 0.19365079365079366
              },
              "scored": 8324,
              "sign_reversed_diagnostic": {
                "auroc": 0.2220501935322762,
                "average_precision_sklearn": 0.022121477706621068,
                "precision_at_capacity": 0,
                "recall_at_capacity": 0
              },
              "unique_predictions": 8324
            },
            "dnabert2-117m-frozen-pair-logreg": {
              "constant_ranking_control": {
                "auroc": 0.5,
                "average_precision_sklearn": 0.037842383469485825,
                "precision_at_capacity": 0.04,
                "recall_at_capacity": 0.012698412698412698
              },
              "original": {
                "auroc": 0.5500324040216661,
                "average_precision_sklearn": 0.04508654312652131,
                "precision_at_capacity": 0.03,
                "recall_at_capacity": 0.009523809523809525
              },
              "scored": 8324,
              "sign_reversed_diagnostic": {
                "auroc": 0.4499675959783339,
                "average_precision_sklearn": 0.0334380195616437,
                "precision_at_capacity": 0.01,
                "recall_at_capacity": 0.0031746031746031746
              },
              "unique_predictions": 8324
            },
            "pangolin-maskFalse": {
              "constant_ranking_control": {
                "auroc": 0.5,
                "average_precision_sklearn": 0.037826767859294064,
                "precision_at_capacity": 0.03,
                "recall_at_capacity": 0.009554140127388535
              },
              "original": {
                "auroc": 0.8756851300560864,
                "average_precision_sklearn": 0.3887617543064248,
                "precision_at_capacity": 0.65,
                "recall_at_capacity": 0.2070063694267516
              },
              "scored": 8301,
              "sign_reversed_diagnostic": {
                "auroc": 0.12431486994391364,
                "average_precision_sklearn": 0.020805299220831536,
                "precision_at_capacity": 0,
                "recall_at_capacity": 0
              },
              "unique_predictions": 86
            },
            "spliceai-1.3.1": {
              "constant_ranking_control": {
                "auroc": 0.5,
                "average_precision_sklearn": 0.03758847937515255,
                "precision_at_capacity": 0.04,
                "recall_at_capacity": 0.012987012987012988
              },
              "original": {
                "auroc": 0.8055241740253153,
                "average_precision_sklearn": 0.2986855472760137,
                "precision_at_capacity": 0.64,
                "recall_at_capacity": 0.2077922077922078
              },
              "scored": 8194,
              "sign_reversed_diagnostic": {
                "auroc": 0.19447582597468474,
                "average_precision_sklearn": 0.021966631459478806,
                "precision_at_capacity": 0.01,
                "recall_at_capacity": 0.003246753246753247
              },
              "unique_predictions": 93
            }
          }
        },
        "operation_id": "agent-1-2-score-information-diagnostics",
        "table_sha256": "c57f9cfe91d02ab54f6bc20934f5b8f678b37c3b13d8d20aa805fa7d73e62c4f"
      },
      "receipt_sha256": "bbfd0de2351d895b84d380c4d3cebcdb1e581d4cf7d5d03ebc45526db97be525",
      "started_at": "2026-09-23T23:12:55.445440Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:12:57.129136Z",
      "id": "attempt-d4d350cdb51be799a77952c8",
      "operation": {
        "expected_observation": "Using baseline-kmer-position-v2 as reference, attenuation or reversal of candidate-minus-reference gaps on identical common scored rows would support a population-composition explanation and weaken attribution to boundary dependence. Persistent gaps would weaken coverage as the main explanation, while providing the matched-population comparison needed to interpret boundary results. Differences between AUROC, AP and capacity metrics would establish metric dependence, not biological superiority.",
        "field": null,
        "id": "agent-2-1-common-row-performance",
        "kind": "paired",
        "methods": [
          "baseline-kmer-position-v2",
          "dnabert2-117m-frozen-pair-logreg",
          "pangolin-maskFalse",
          "spliceai-1.3.1"
        ],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "dc65f50a8617ad1c2d3cf4e75e01dcb67574076abbf8094a7350ed9b53842e42",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "paired",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "4fca3c60f42b6f2d4259b0aa2c0c339d05d32b5db9f191159fa1b41ac9a97e64",
        "numerical": {
          "pairs": [
            {
              "candidate": "dnabert2-117m-frozen-pair-logreg",
              "candidate_metrics": {
                "auroc": 0.5500324040216661,
                "average_precision_sklearn": 0.04508654312652131,
                "precision_at_capacity": 0.03,
                "recall_at_capacity": 0.009523809523809525
              },
              "candidate_minus_reference": {
                "auroc": -0.22791740244605774,
                "average_precision_sklearn": -0.2413302028324024,
                "precision_at_capacity": -0.58,
                "recall_at_capacity": -0.18412698412698414
              },
              "common_n": 8324,
              "original_n": 8324,
              "reference": "baseline-kmer-position-v2",
              "reference_metrics": {
                "auroc": 0.7779498064677238,
                "average_precision_sklearn": 0.2864167459589237,
                "precision_at_capacity": 0.61,
                "recall_at_capacity": 0.19365079365079366
              }
            },
            {
              "candidate": "pangolin-maskFalse",
              "candidate_metrics": {
                "auroc": 0.8756851300560864,
                "average_precision_sklearn": 0.3887617543064248,
                "precision_at_capacity": 0.65,
                "recall_at_capacity": 0.2070063694267516
              },
              "candidate_minus_reference": {
                "auroc": 0.0979898465579816,
                "average_precision_sklearn": 0.10181178069351354,
                "precision_at_capacity": 0.040000000000000036,
                "recall_at_capacity": 0.01273885350318471
              },
              "common_n": 8301,
              "original_n": 8324,
              "reference": "baseline-kmer-position-v2",
              "reference_metrics": {
                "auroc": 0.7776952834981048,
                "average_precision_sklearn": 0.28694997361291125,
                "precision_at_capacity": 0.61,
                "recall_at_capacity": 0.1942675159235669
              }
            },
            {
              "candidate": "spliceai-1.3.1",
              "candidate_metrics": {
                "auroc": 0.8055241740253153,
                "average_precision_sklearn": 0.2986855472760137,
                "precision_at_capacity": 0.64,
                "recall_at_capacity": 0.2077922077922078
              },
              "candidate_minus_reference": {
                "auroc": 0.02840723820941926,
                "average_precision_sklearn": 0.009023934617023666,
                "precision_at_capacity": 0.030000000000000027,
                "recall_at_capacity": 0.009740259740259744
              },
              "common_n": 8194,
              "original_n": 8324,
              "reference": "baseline-kmer-position-v2",
              "reference_metrics": {
                "auroc": 0.777116935815896,
                "average_precision_sklearn": 0.28966161265899004,
                "precision_at_capacity": 0.61,
                "recall_at_capacity": 0.19805194805194806
              }
            },
            {
              "candidate": "pangolin-maskFalse",
              "candidate_metrics": {
                "auroc": 0.8756851300560864,
                "average_precision_sklearn": 0.3887617543064248,
                "precision_at_capacity": 0.65,
                "recall_at_capacity": 0.2070063694267516
              },
              "candidate_minus_reference": {
                "auroc": 0.32534775857902853,
                "average_precision_sklearn": 0.3436552116516486,
                "precision_at_capacity": 0.62,
                "recall_at_capacity": 0.19745222929936307
              },
              "common_n": 8301,
              "original_n": 8324,
              "reference": "dnabert2-117m-frozen-pair-logreg",
              "reference_metrics": {
                "auroc": 0.5503373714770579,
                "average_precision_sklearn": 0.045106542654776205,
                "precision_at_capacity": 0.03,
                "recall_at_capacity": 0.009554140127388535
              }
            },
            {
              "candidate": "spliceai-1.3.1",
              "candidate_metrics": {
                "auroc": 0.8055241740253153,
                "average_precision_sklearn": 0.2986855472760137,
                "precision_at_capacity": 0.64,
                "recall_at_capacity": 0.2077922077922078
              },
              "candidate_minus_reference": {
                "auroc": 0.2517897078827842,
                "average_precision_sklearn": 0.2534673437329571,
                "precision_at_capacity": 0.61,
                "recall_at_capacity": 0.19805194805194806
              },
              "common_n": 8194,
              "original_n": 8324,
              "reference": "dnabert2-117m-frozen-pair-logreg",
              "reference_metrics": {
                "auroc": 0.5537344661425311,
                "average_precision_sklearn": 0.04521820354305664,
                "precision_at_capacity": 0.03,
                "recall_at_capacity": 0.00974025974025974
              }
            },
            {
              "candidate": "spliceai-1.3.1",
              "candidate_metrics": {
                "auroc": 0.8055241740253153,
                "average_precision_sklearn": 0.2986855472760137,
                "precision_at_capacity": 0.64,
                "recall_at_capacity": 0.2077922077922078
              },
              "candidate_minus_reference": {
                "auroc": -0.06981919298049133,
                "average_precision_sklearn": -0.09213206420632947,
                "precision_at_capacity": -0.010000000000000009,
                "recall_at_capacity": -0.00324675324675322
              },
              "common_n": 8194,
              "original_n": 8324,
              "reference": "pangolin-maskFalse",
              "reference_metrics": {
                "auroc": 0.8753433670058066,
                "average_precision_sklearn": 0.3908176114823432,
                "precision_at_capacity": 0.65,
                "recall_at_capacity": 0.21103896103896103
              }
            }
          ]
        },
        "operation_id": "agent-2-1-common-row-performance",
        "table_sha256": "c57f9cfe91d02ab54f6bc20934f5b8f678b37c3b13d8d20aa805fa7d73e62c4f"
      },
      "receipt_sha256": "ee410d8ec1f1e9ab9b251e8b7e9172817332b00f633caae3a129c417e9cece91",
      "started_at": "2026-09-23T23:12:56.335200Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:12:57.929701Z",
      "id": "attempt-8ec25ed12c3900235fded55d",
      "operation": {
        "expected_observation": "Variation in relative performance across predeclared boundary bands would support boundary-associated heterogeneity; similar gaps across adequately populated bands would weaken it. Interpret AP alongside outcome means and per-method scored counts. Unequal coverage or single-class bands limit comparisons. These metrics do not directly measure variant-level prediction disagreement.",
        "field": "boundary_band",
        "id": "agent-2-2-boundary-band-performance",
        "kind": "subgroups",
        "methods": [
          "baseline-kmer-position-v2",
          "dnabert2-117m-frozen-pair-logreg",
          "pangolin-maskFalse",
          "spliceai-1.3.1"
        ],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "dc65f50a8617ad1c2d3cf4e75e01dcb67574076abbf8094a7350ed9b53842e42",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "subgroups",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "4fca3c60f42b6f2d4259b0aa2c0c339d05d32b5db9f191159fa1b41ac9a97e64",
        "numerical": {
          "binning": {
            "edges": [],
            "rule": "predeclared annotation categories"
          },
          "field": "boundary_band",
          "subgroups": [
            {
              "label": "0-2",
              "methods": {
                "baseline-kmer-position-v2": {
                  "metrics": {
                    "auroc": 0.8377902885292048,
                    "average_precision_sklearn": 0.5183134824574966,
                    "precision_at_capacity": 0.29,
                    "recall_at_capacity": 0.6904761904761905
                  },
                  "scored": 448
                },
                "dnabert2-117m-frozen-pair-logreg": {
                  "metrics": {
                    "auroc": 0.6020408163265306,
                    "average_precision_sklearn": 0.13520653880519168,
                    "precision_at_capacity": 0.16,
                    "recall_at_capacity": 0.38095238095238093
                  },
                  "scored": 448
                },
                "pangolin-maskFalse": {
                  "metrics": {
                    "auroc": 0.9238212526389866,
                    "average_precision_sklearn": 0.6453656386173421,
                    "precision_at_capacity": 0.34,
                    "recall_at_capacity": 0.8095238095238095
                  },
                  "scored": 448
                },
                "spliceai-1.3.1": {
                  "metrics": {
                    "auroc": 0.9017716296565951,
                    "average_precision_sklearn": 0.5857307167402802,
                    "precision_at_capacity": 0.31,
                    "recall_at_capacity": 0.7560975609756098
                  },
                  "scored": 443
                }
              },
              "n": 448,
              "outcome_mean": 0.09375
            },
            {
              "label": "11-30",
              "methods": {
                "baseline-kmer-position-v2": {
                  "metrics": {
                    "auroc": 0.7642881220887472,
                    "average_precision_sklearn": 0.26347426873322705,
                    "precision_at_capacity": 0.38,
                    "recall_at_capacity": 0.2375
                  },
                  "scored": 4239
                },
                "dnabert2-117m-frozen-pair-logreg": {
                  "metrics": {
                    "auroc": 0.5168101863201765,
                    "average_precision_sklearn": 0.04278432946451933,
                    "precision_at_capacity": 0.04,
                    "recall_at_capacity": 0.025
                  },
                  "scored": 4239
                },
                "pangolin-maskFalse": {
                  "metrics": {
                    "auroc": 0.868495158918307,
                    "average_precision_sklearn": 0.36263307502559783,
                    "precision_at_capacity": 0.48,
                    "recall_at_capacity": 0.3018867924528302
                  },
                  "scored": 4226
                },
                "spliceai-1.3.1": {
                  "metrics": {
                    "auroc": 0.7848152525533147,
                    "average_precision_sklearn": 0.28420482071335973,
                    "precision_at_capacity": 0.42,
                    "recall_at_capacity": 0.2641509433962264
                  },
                  "scored": 4166
                }
              },
              "n": 4239,
              "outcome_mean": 0.0377447511205473
            },
            {
              "label": "3-10",
              "methods": {
                "baseline-kmer-position-v2": {
                  "metrics": {
                    "auroc": 0.7368659316038642,
                    "average_precision_sklearn": 0.2818487669278446,
                    "precision_at_capacity": 0.3,
                    "recall_at_capacity": 0.37037037037037035
                  },
                  "scored": 1690
                },
                "dnabert2-117m-frozen-pair-logreg": {
                  "metrics": {
                    "auroc": 0.591840649433357,
                    "average_precision_sklearn": 0.0628302214735115,
                    "precision_at_capacity": 0.05,
                    "recall_at_capacity": 0.06172839506172839
                  },
                  "scored": 1690
                },
                "pangolin-maskFalse": {
                  "metrics": {
                    "auroc": 0.8594054074843276,
                    "average_precision_sklearn": 0.384449295653607,
                    "precision_at_capacity": 0.43,
                    "recall_at_capacity": 0.5308641975308642
                  },
                  "scored": 1686
                },
                "spliceai-1.3.1": {
                  "metrics": {
                    "auroc": 0.7977561961250794,
                    "average_precision_sklearn": 0.27226583608733973,
                    "precision_at_capacity": 0.32,
                    "recall_at_capacity": 0.4155844155844156
                  },
                  "scored": 1671
                }
              },
              "n": 1690,
              "outcome_mean": 0.04792899408284024
            },
            {
              "label": ">30",
              "methods": {
                "baseline-kmer-position-v2": {
                  "metrics": {
                    "auroc": 0.7594484334203656,
                    "average_precision_sklearn": 0.043619720237491,
                    "precision_at_capacity": 0.04,
                    "recall_at_capacity": 0.125
                  },
                  "scored": 1947
                },
                "dnabert2-117m-frozen-pair-logreg": {
                  "metrics": {
                    "auroc": 0.5490535248041775,
                    "average_precision_sklearn": 0.02098274499062513,
                    "precision_at_capacity": 0.02,
                    "recall_at_capacity": 0.0625
                  },
                  "scored": 1947
                },
                "pangolin-maskFalse": {
                  "metrics": {
                    "auroc": 0.8376849790466212,
                    "average_precision_sklearn": 0.1486754959430206,
                    "precision_at_capacity": 0.15,
                    "recall_at_capacity": 0.46875
                  },
                  "scored": 1941
                },
                "spliceai-1.3.1": {
                  "metrics": {
                    "auroc": 0.7437513919106435,
                    "average_precision_sklearn": 0.07335603629452685,
                    "precision_at_capacity": 0.09,
                    "recall_at_capacity": 0.2903225806451613
                  },
                  "scored": 1914
                }
              },
              "n": 1947,
              "outcome_mean": 0.016435541859270673
            }
          ]
        },
        "operation_id": "agent-2-2-boundary-band-performance",
        "table_sha256": "c57f9cfe91d02ab54f6bc20934f5b8f678b37c3b13d8d20aa805fa7d73e62c4f"
      },
      "receipt_sha256": "fa138be4e492f3aa6f1ebeb7dcb5ea8fda904e97ee6b6473f34d0b8a5c1f9965",
      "started_at": "2026-09-23T23:12:57.129940Z",
      "status": "completed"
    }
  ],
  "catalogue_release_id": "2026-09-20-370b30415b09",
  "claim_level": "exploratory",
  "created_at": "2026-09-23T23:00:33.952383Z",
  "execution": {
    "campaign_id": "campaign-fd29c4413eaf4e24",
    "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
    "codex_calls": 2,
    "reasoning": [
      {
        "cli_version": "codex-cli 0.155.0-alpha.9.2",
        "model": "gpt-6-astra",
        "role": "planner",
        "usage": {
          "cache_write_input_tokens": 0,
          "cached_input_tokens": 0,
          "input_tokens": 13261,
          "output_tokens": 7764,
          "reasoning_output_tokens": 6638
        }
      },
      {
        "cli_version": "codex-cli 0.155.0-alpha.9.2",
        "model": "gpt-6-astra",
        "role": "critic",
        "usage": {
          "cache_write_input_tokens": 0,
          "cached_input_tokens": 0,
          "input_tokens": 22732,
          "output_tokens": 4812,
          "reasoning_output_tokens": 4142
        }
      }
    ]
  },
  "findings": [
    "Verification reported 8,324 unique IDs and all artifact checks passed; all 16 metric replays matched within 1e-9. No declared hash or replay mismatch supports an evidence-integrity explanation.",
    "dnabert2-117m-frozen-pair-logreg and baseline-kmer-position-v2 both score all 8,324 rows, so missingness cannot explain their average_precision_sklearn gap of -0.24133. The former has 8,324 distinct predictions, contradicting constant-score collapse. Its average_precision_sklearn is 0.04509 versus 0.03784 for the constant-ranking control; reversing scores lowers auroc from 0.55003 to 0.44997. Weak observed ranking remains; simple sign inversion is unsupported.",
    "On pairwise common rows, pangolin-maskFalse and spliceai-1.3.1 retain average_precision_sklearn gains over baseline-kmer-position-v2 of 0.10181 and 0.00902. Unequal coverage is therefore insufficient to explain these observed advantages.",
    "Denominators explain one apparent ranking reversal: unmatched recall_at_capacity favors spliceai-1.3.1 over pangolin-maskFalse, 0.20779 versus 0.20701. On their 8,194 common rows, pangolin-maskFalse leads, 0.21104 versus 0.20779.",
    "Reported boundary_band performance varies by metric: spliceai-1.3.1 exceeds baseline-kmer-position-v2 in auroc at 0-2 (0.90177 versus 0.83779), but falls below at >30 (0.74375 versus 0.75945). At >30 its average_precision_sklearn remains higher (0.07336 versus 0.04362).",
    "pangolin-maskFalse retains higher auroc and average_precision_sklearn than baseline-kmer-position-v2 in every boundary_band, contradicting an advantage restricted to the nearest band. Outcome prevalence falls from 0.09375 at 0-2 to 0.01644 at >30, limiting interpretation of average_precision_sklearn differences across bands."
  ],
  "id": "mfass-v2-discrepancies-campaign-fd29c4413eaf4e24",
  "limitations": [
    "Existing test outcomes have been inspected; all new subgroup findings are exploratory.",
    "Specialists use different genomic context and annotation releases; model-only attribution is unsupported.",
    "Boundary bands are symmetric distances, not canonical dinucleotide annotations; baseline uses distance features.",
    "The historical v1 sequence-orientation error is already corrected and is not a new discovery.",
    "Source assay data have no declared redistribution licence; obtain original tables from KosuriLab/MFASS.",
    "Exploratory analysis of exposed outcomes; no independent validation or novelty claim.",
    "AI-assisted interpretation requires human scientific review before publication.",
    "All intervals are descriptive and unadjusted; unsuccessful tests remain in the report.",
    "Aggregate performance does not directly measure prediction disagreement. Boundary-associated performance variation is descriptive; replicate_gap_band remains untested and gene/exon concentration remains unresolved.",
    "Paired comparisons use pair-specific intersections, and subgroup metrics use each method's available predictions. Whole-population matching does not establish matched comparisons within boundary_band.",
    "Requested tool (unavailable): connected_exon_gene_group_completeness_and_concentration_audit. No numerical receipt establishes group completeness or documented independence sufficient for bootstrap.",
    "Hash verification and metric replay do not establish correct upstream preparation. These receipts cannot isolate model effects from input context, annotation choices or distance features, or establish canonical dinucleotide effects.",
    "Previously exposed data and descriptive comparisons support no independent-validation, discovery, mechanism or statistical-significance claim. No follow-up rounds remain."
  ],
  "manifest_id": "mfass-v2-discrepancies",
  "outcome": "inconclusive",
  "plan_sha256": "30ad11711675d2c0b9c20d027589645322d73fa15ea535e39ff71711886cc66e",
  "question": "Does model disagreement depend on exon boundary distance, assay replicate agreement or gene/exon concentration?",
  "review": {
    "method": "ai_assisted",
    "status": "pending"
  },
  "schema_version": "1.0",
  "specs": [
    {
      "budget": {
        "campaign_seconds": 3600,
        "codex_calls": 8,
        "codex_seconds": 900,
        "experiment_seconds": 300,
        "followup_rounds": 0,
        "memory_bytes": 2147483648,
        "workspace_bytes": 1073741824
      },
      "catalogue_release_id": "2026-09-20-370b30415b09",
      "created_at": "2026-09-23T23:12:52.939071Z",
      "evidence": [
        {
          "format": "tsv",
          "id": "mfass-v2-discrepancies-cohort",
          "role": "cohort",
          "sha256": "389702ff4c647d7ce10a90092a6fa811ae777d15997baf39ce9aae0346247bd0",
          "uri": null
        },
        {
          "format": "txt",
          "id": "mfass-v2-discrepancies-outcomes",
          "role": "outcomes",
          "sha256": "a637ca0e307e66ff48811ec7efa22b9ce453bc7883b04f0cacb867f7283132d8",
          "uri": "https://raw.githubusercontent.com/KosuriLab/MFASS/master/processed_data/snv/snv_data_clean.txt"
        },
        {
          "format": "txt",
          "id": "mfass-v2-discrepancies-annotations",
          "role": "annotations",
          "sha256": "71a857fe647c4e68acbb41ca61e959c47e1176de89b1442bd6ca1772aa60d5a1",
          "uri": "https://raw.githubusercontent.com/KosuriLab/MFASS/master/processed_data/snv/snv_func_annot.txt"
        },
        {
          "format": "tsv",
          "id": "mfass-v2-discrepancies-split",
          "role": "split",
          "sha256": "999ebcb7e63a5c5eaa8780fa468e59ac1f934260ad50102814174c396317f052",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/splits/split-v2.tsv"
        },
        {
          "format": "npy",
          "id": "mfass-v2-discrepancies-baseline-kmer-position-v2-scores",
          "role": "baseline-kmer-position-v2-scores",
          "sha256": "ed0ebb3d74183deb1fb475d0e706c0b1211fd6ab9d20449c2c194e57f912d862",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/baseline-kmer-position-v2.scores.npy"
        },
        {
          "format": "tsv",
          "id": "mfass-v2-discrepancies-baseline-kmer-position-v2-predictions",
          "role": "baseline-kmer-position-v2-predictions",
          "sha256": "2f3117c225a8da9ea737abfa2d0f7e1dec97696ae4bacd263864bec9d7f923eb",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/baseline-kmer-position-v2.predictions.tsv"
        },
        {
          "format": "json",
          "id": "mfass-v2-discrepancies-baseline-kmer-position-v2-report",
          "role": "baseline-kmer-position-v2-report",
          "sha256": "9a0b78674cc714177fec6e8c487588d6186d4fe93d6d7dbcb48bed6858885e15",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/baseline-kmer-position-v2.json"
        },
        {
          "format": "npy",
          "id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-scores",
          "role": "dnabert2-117m-frozen-pair-logreg-scores",
          "sha256": "e6b019babf0b3fb9ec05d4261a9383c0c896183a76c3b31634564c7a63d01105",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.scores.npy"
        },
        {
          "format": "tsv",
          "id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-predictions",
          "role": "dnabert2-117m-frozen-pair-logreg-predictions",
          "sha256": "3abded2932366e6e2c8d6e0da5836c4240239372758b3c68cd5f19635b8cee11",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.predictions.tsv"
        },
        {
          "format": "json",
          "id": "mfass-v2-discrepancies-dnabert2-117m-frozen-pair-logreg-report",
          "role": "dnabert2-117m-frozen-pair-logreg-report",
          "sha256": "60b28541853de349f878d6d1ccbcbe7dfd21b69db976e9c01f60e88bdba0f2a1",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.json"
        },
        {
          "format": "tsv",
          "id": "mfass-v2-discrepancies-spliceai-1.3.1-predictions",
          "role": "spliceai-1.3.1-predictions",
          "sha256": "b39df773067e4ecd862980f177da69d6117c75c93c754f4cbdd7a77bd419851f",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/spliceai-1.3.1.predictions.tsv"
        },
        {
          "format": "json",
          "id": "mfass-v2-discrepancies-spliceai-1.3.1-report",
          "role": "spliceai-1.3.1-report",
          "sha256": "6d5c59eb0fd60d95064d97e331c9cdb12b2a34db7dc509ff73dfb9502ddf02dd",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/spliceai-1.3.1.json"
        },
        {
          "format": "tsv",
          "id": "mfass-v2-discrepancies-pangolin-maskFalse-predictions",
          "role": "pangolin-maskFalse-predictions",
          "sha256": "faa7d4cb3728111f1c5a8e8be6cd2c5958c4cced7c7a366ec96171d03b855ad6",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/pangolin-maskFalse.predictions.tsv"
        },
        {
          "format": "json",
          "id": "mfass-v2-discrepancies-pangolin-maskFalse-report",
          "role": "pangolin-maskFalse-report",
          "sha256": "bb0bb6732808699e54938233df1835dfc1f775f33ba7d6acd916e53d588a3c44",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/benchmarks/mfass/results/pangolin-maskFalse.json"
        },
        {
          "format": "json",
          "id": "mfass-v2-discrepancies-table",
          "role": "table",
          "sha256": "c57f9cfe91d02ab54f6bc20934f5b8f678b37c3b13d8d20aa805fa7d73e62c4f",
          "uri": null
        },
        {
          "format": "json",
          "id": "mfass-v2-discrepancies-receipt",
          "role": "receipt",
          "sha256": "a9fa54bcf99457f9b23424ad656a952eb262b376845327e267dc035cf85a012e",
          "uri": null
        },
        {
          "format": "py",
          "id": "mfass-v2-discrepancies-preparation_code",
          "role": "preparation_code",
          "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
          "uri": null
        },
        {
          "format": "lock",
          "id": "mfass-v2-discrepancies-environment",
          "role": "environment",
          "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
          "uri": null
        }
      ],
      "exposure": {
        "independent_validation": false,
        "previously_exposed": true,
        "usage": "exploration"
      },
      "id": "mfass-v2-discrepancies-round-0",
      "manifest_id": "mfass-v2-discrepancies",
      "manifest_sha256": "4fca3c60f42b6f2d4259b0aa2c0c339d05d32b5db9f191159fa1b41ac9a97e64",
      "permitted_actions": [
        "verify",
        "replay",
        "coverage",
        "paired",
        "subgroups",
        "bootstrap",
        "sensitivity",
        "local_recipe"
      ],
      "plan": {
        "hypotheses": [
          {
            "explanation": "The discrepancy may arise from stale evidence, metric calculation or an inadequate trivial control.",
            "id": "evidence-and-method",
            "tests": [
              {
                "expected_observation": "All available evidence matches its declared byte hash.",
                "field": null,
                "id": "verify-evidence",
                "kind": "verify",
                "methods": [],
                "metric": null,
                "recipe": null
              },
              {
                "expected_observation": "Recorded metrics replay within the stated tolerance.",
                "field": null,
                "id": "replay-metrics",
                "kind": "replay",
                "methods": [],
                "metric": null,
                "recipe": null
              }
            ]
          },
          {
            "explanation": "Unequal scored populations or weak score information may contribute to the supplied performance differences. The AP targets are 0.04508654312652131 for dnabert2-117m-frozen-pair-logreg and 0.2864167459589237 for baseline-kmer-position-v2; these targets do not establish missingness, score collapse or a direction error.",
            "id": "agent-1-coverage-or-score-information",
            "tests": [
              {
                "expected_observation": "Different missing-prediction counts would support a denominator concern requiring the common-row comparison. Complete coverage for every method would contradict missingness as an explanation. Coverage alone cannot establish selection bias.",
                "field": null,
                "id": "agent-1-1-whole-population-coverage",
                "kind": "coverage",
                "methods": [
                  "baseline-kmer-position-v2",
                  "dnabert2-117m-frozen-pair-logreg",
                  "pangolin-maskFalse",
                  "spliceai-1.3.1"
                ],
                "metric": null,
                "recipe": null
              },
              {
                "expected_observation": "Few unique scores and performance resembling the registered constant-ranking control would support score collapse or weak ranking information. Many unique scores would contradict collapse, though not weak information. Better reversed-score performance would flag a semantics question without establishing an error or authorizing a direction change.",
                "field": null,
                "id": "agent-1-2-score-information-diagnostics",
                "kind": "sensitivity",
                "methods": [
                  "baseline-kmer-position-v2",
                  "dnabert2-117m-frozen-pair-logreg",
                  "pangolin-maskFalse",
                  "spliceai-1.3.1"
                ],
                "metric": null,
                "recipe": null
              }
            ]
          },
          {
            "explanation": "Relative performance may vary with boundary distance. This is an exploratory association hypothesis; distance bands cannot establish canonical dinucleotide effects, and differences in model inputs or distance features prevent model-only attribution.",
            "id": "agent-2-boundary-dependent-performance",
            "tests": [
              {
                "expected_observation": "Using baseline-kmer-position-v2 as reference, attenuation or reversal of candidate-minus-reference gaps on identical common scored rows would support a population-composition explanation and weaken attribution to boundary dependence. Persistent gaps would weaken coverage as the main explanation, while providing the matched-population comparison needed to interpret boundary results. Differences between AUROC, AP and capacity metrics would establish metric dependence, not biological superiority.",
                "field": null,
                "id": "agent-2-1-common-row-performance",
                "kind": "paired",
                "methods": [
                  "baseline-kmer-position-v2",
                  "dnabert2-117m-frozen-pair-logreg",
                  "pangolin-maskFalse",
                  "spliceai-1.3.1"
                ],
                "metric": null,
                "recipe": null
              },
              {
                "expected_observation": "Variation in relative performance across predeclared boundary bands would support boundary-associated heterogeneity; similar gaps across adequately populated bands would weaken it. Interpret AP alongside outcome means and per-method scored counts. Unequal coverage or single-class bands limit comparisons. These metrics do not directly measure variant-level prediction disagreement.",
                "field": "boundary_band",
                "id": "agent-2-2-boundary-band-performance",
                "kind": "subgroups",
                "methods": [
                  "baseline-kmer-position-v2",
                  "dnabert2-117m-frozen-pair-logreg",
                  "pangolin-maskFalse",
                  "spliceai-1.3.1"
                ],
                "metric": null,
                "recipe": null
              }
            ]
          }
        ],
        "multiple_testing": "descriptive_only",
        "requested_tools": [
          "connected_exon_gene_group_completeness_and_concentration_audit"
        ],
        "stopping_rule": "Proceed only after the supervisor's verification and metric replay succeed; do not repeat them or fit another core baseline. Execute coverage and sensitivity before paired and subgroup analyses. Retain predeclared subgroup labels; numerical annotations use only the registered fixed annotation-only quartiles. Do not optimize cutoffs, reverse the declared score direction or change protocol capacity. Stop after the four additional tests and preserve contradictory or inconclusive results. Complete connected exon/gene group labels are not established by a numerical receipt, so bootstrap is not currently eligible. Gene/exon concentration cannot be tested by the registered subgroup fields; assay replicate agreement also remains untested in this plan. All findings remain exploratory on exposed public data, with no significance, mechanism or independent-validation claims. Request a follow-up only when a new registered test would materially distinguish a remaining explanation."
      },
      "question": "Does model disagreement depend on exon boundary distance, assay replicate agreement or gene/exon concentration?",
      "round": 0,
      "schema_version": "1.0"
    }
  ],
  "status": "completed",
  "title": "MFASS v2: model disagreement"
}
