{
  "artifacts": [
    {
      "format": "json",
      "id": "flip2-discrepancies-prepared",
      "role": "prepared",
      "semantic_sha256": "f417d43497e0c8c453bd7c66cfea05959b4d6d47462283096e42a2dae8b72419",
      "sha256": "df996aee84bd5ed3e6381f9f6db864a377ab0b4809a9a48ff146d06a3dc069df",
      "uri": null
    },
    {
      "format": "json",
      "id": "flip2-discrepancies-flip2-composition-predictions",
      "role": "flip2-composition-predictions",
      "sha256": "72ea3c506b8645ae6020207f904fbd5c0b09549cd125fcdcabfb487e68681b9c",
      "uri": null
    },
    {
      "format": "json",
      "id": "flip2-discrepancies-flip2-composition-report",
      "role": "flip2-composition-report",
      "sha256": "abd08d3f493428d06f43740c94e268575d755560e72120c73b294127acb395e6",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-composition/report.json"
    },
    {
      "format": "json",
      "id": "flip2-discrepancies-flip2-composition-source",
      "role": "flip2-composition-source",
      "sha256": "7157313eab75ff8f93b934ac186852c8c82f8bb885617cab70c89427c631c1e4",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-composition/source.json"
    },
    {
      "format": "json",
      "id": "flip2-discrepancies-flip2-train-mean-predictions",
      "role": "flip2-train-mean-predictions",
      "sha256": "599dca4e6fa643a7cdfc96ac339033f61e5846ce225af17ac2ac3917ecc14fac",
      "uri": null
    },
    {
      "format": "json",
      "id": "flip2-discrepancies-flip2-train-mean-report",
      "role": "flip2-train-mean-report",
      "sha256": "eb89260f6e138bb9ae2fdf3b882a87fdf8aeabd54a0ac6f17a73f5671c96b4d8",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-train-mean/report.json"
    },
    {
      "format": "json",
      "id": "flip2-discrepancies-flip2-train-mean-source",
      "role": "flip2-train-mean-source",
      "sha256": "8bd712289f136edbfde3fa885166ef85e4f6fe3c4c437810c24dadb87107cc3f",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-train-mean/source.json"
    },
    {
      "format": "py",
      "id": "flip2-discrepancies-recipe_code",
      "role": "recipe_code",
      "sha256": "b3407940ca4c8a1cdcf3a3bb3849c3e26d7f85272b2ee63087307378ffaa79dc",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/packages/rewirebench/src/rewirebench/adapters/sequence.py"
    },
    {
      "format": "json",
      "id": "flip2-discrepancies-table",
      "role": "table",
      "sha256": "8562d4891270a73026fc7852c05f41206ad52e8fcae1d706d83ec57572c7c949",
      "uri": null
    },
    {
      "format": "json",
      "id": "flip2-discrepancies-receipt",
      "role": "receipt",
      "sha256": "68631e653cd5aa9047b3f4e88cbb3fec94630847c1b73fad8de1018423d81fac",
      "uri": null
    },
    {
      "format": "py",
      "id": "flip2-discrepancies-preparation_code",
      "role": "preparation_code",
      "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
      "uri": null
    },
    {
      "format": "lock",
      "id": "flip2-discrepancies-environment",
      "role": "environment",
      "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
      "uri": null
    }
  ],
  "attempts": [
    {
      "error": null,
      "finished_at": "2026-09-23T23:07:41.510732Z",
      "id": "attempt-18396ae1d3116e62b4e6479c",
      "operation": {
        "expected_observation": "All available evidence matches its declared byte hash.",
        "field": null,
        "id": "verify-evidence",
        "kind": "verify",
        "methods": [],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "9a0788c6679363b7e33109d0a4fbf09381aa7d1e5646384286893d02fea61fe3",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "verify",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "fc60f1bff46326bd0410063f51c3f4fcec8f0c68a35c8ac11a2938674ef7468e",
        "numerical": {
          "checks": [
            {
              "artifact_id": "flip2-discrepancies-prepared",
              "status": "passed"
            },
            {
              "artifact_id": "flip2-discrepancies-flip2-composition-predictions",
              "status": "passed"
            },
            {
              "artifact_id": "flip2-discrepancies-flip2-composition-report",
              "status": "passed"
            },
            {
              "artifact_id": "flip2-discrepancies-flip2-composition-source",
              "status": "passed"
            },
            {
              "artifact_id": "flip2-discrepancies-flip2-train-mean-predictions",
              "status": "passed"
            },
            {
              "artifact_id": "flip2-discrepancies-flip2-train-mean-report",
              "status": "passed"
            },
            {
              "artifact_id": "flip2-discrepancies-flip2-train-mean-source",
              "status": "passed"
            },
            {
              "artifact_id": "flip2-discrepancies-recipe_code",
              "status": "passed"
            },
            {
              "artifact_id": "flip2-discrepancies-table",
              "status": "passed"
            },
            {
              "artifact_id": "flip2-discrepancies-receipt",
              "status": "passed"
            },
            {
              "artifact_id": "flip2-discrepancies-preparation_code",
              "status": "passed"
            },
            {
              "artifact_id": "flip2-discrepancies-environment",
              "status": "passed"
            }
          ],
          "identifiers_unique": true,
          "methods": [
            "flip2-composition",
            "flip2-train-mean"
          ],
          "rows": 184
        },
        "operation_id": "verify-evidence",
        "table_sha256": "8562d4891270a73026fc7852c05f41206ad52e8fcae1d706d83ec57572c7c949"
      },
      "receipt_sha256": "514ee4705e11f0884808ce032222ae68765b8e2d075b9d4785c3049de51efcd7",
      "started_at": "2026-09-23T23:07:40.722247Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:07:42.311552Z",
      "id": "attempt-19d1a20e000ed9e604b3c827",
      "operation": {
        "expected_observation": "Recorded metrics replay within the stated tolerance.",
        "field": null,
        "id": "replay-metrics",
        "kind": "replay",
        "methods": [],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "9a0788c6679363b7e33109d0a4fbf09381aa7d1e5646384286893d02fea61fe3",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "replay",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "fc60f1bff46326bd0410063f51c3f4fcec8f0c68a35c8ac11a2938674ef7468e",
        "numerical": {
          "checks": [
            {
              "actual": 0.954819941821582,
              "expected": 0.954819941821582,
              "method": "flip2-composition",
              "metric": "ndcg",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.41818225155801514,
              "expected": 0.41818225155801514,
              "method": "flip2-composition",
              "metric": "spearman",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.9206667522227658,
              "expected": 0.9206667522227658,
              "method": "flip2-train-mean",
              "metric": "ndcg",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": null,
              "expected": null,
              "method": "flip2-train-mean",
              "metric": "spearman",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            }
          ],
          "coverage": {
            "flip2-composition": {
              "denominator": 184,
              "missing": 0,
              "scored": 184
            },
            "flip2-train-mean": {
              "denominator": 184,
              "missing": 0,
              "scored": 184
            }
          },
          "metrics": {
            "flip2-composition": {
              "mse": 1042.666651290419,
              "ndcg": 0.954819941821582,
              "pearson": 0.36211321993490453,
              "spearman": 0.41818225155801514
            },
            "flip2-train-mean": {
              "mse": 1056.3941660479575,
              "ndcg": 0.9206667522227658,
              "pearson": null,
              "spearman": null
            }
          }
        },
        "operation_id": "replay-metrics",
        "table_sha256": "8562d4891270a73026fc7852c05f41206ad52e8fcae1d706d83ec57572c7c949"
      },
      "receipt_sha256": "286f5c40f7effcb1df20d03a0a74025b2a50112742751c6c8133e928f2fddbb8",
      "started_at": "2026-09-23T23:07:41.511329Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:07:43.216777Z",
      "id": "attempt-fce0530a56610e3c80ab3717",
      "operation": {
        "expected_observation": "The SDK trains only on training labels and reproduces the constant control.",
        "field": null,
        "id": "local-train-mean",
        "kind": "local_recipe",
        "methods": [],
        "metric": null,
        "recipe": "sdk:train-mean-v1"
      },
      "plan_sha256": "9a0788c6679363b7e33109d0a4fbf09381aa7d1e5646384286893d02fea61fe3",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "local_recipe",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "fc60f1bff46326bd0410063f51c3f4fcec8f0c68a35c8ac11a2938674ef7468e",
        "numerical": {
          "coverage": {
            "denominator": 184,
            "scored": 184,
            "unscored": 0
          },
          "fitting": "train_only",
          "metrics": {
            "n": 184,
            "ndcg": 0.9206667522227658,
            "spearman": null
          },
          "prepared_sha256": "f417d43497e0c8c453bd7c66cfea05959b4d6d47462283096e42a2dae8b72419",
          "protocol_id": "flip2-fitness-v1",
          "recipe": "sdk:train-mean-v1",
          "report_sha256": "498eaa21e34598198f1579c505f8914e603f62da41c7e6e1182d930ee96924a9",
          "runtime": {
            "adapter_sha256": "b3407940ca4c8a1cdcf3a3bb3849c3e26d7f85272b2ee63087307378ffaa79dc",
            "dependencies": {
              "numpy": "1.26.4",
              "scikit-learn": "1.9.1",
              "scipy": "1.17.1"
            },
            "platform": "Darwin",
            "python": "3.11.13",
            "runner_code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596"
          },
          "scope": "full"
        },
        "operation_id": "local-train-mean",
        "table_sha256": "8562d4891270a73026fc7852c05f41206ad52e8fcae1d706d83ec57572c7c949"
      },
      "receipt_sha256": "2d9316cce4098f894d83a8cf23c19f0538df8c485c1c1a1c98aab4b41a9edd64",
      "started_at": "2026-09-23T23:07:42.312230Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:07:43.994846Z",
      "id": "attempt-2496c45b0a6f24681d4f0d24",
      "operation": {
        "expected_observation": "One unique training-mean score across multiple scored rows, undefined Spearman and unchanged results after score reversal would support score degeneracy. High NDCG for the constant-ranking control would demonstrate that high absolute NDCG requires no discriminative scores here. Multiple unique training-mean scores would contradict strict constancy. Similar composition NDCG despite reversed Spearman would support limited NDCG sensitivity on these outcomes. Reversal remains diagnostic and does not change the declared direction.",
        "field": null,
        "id": "agent-1-1-score-and-ranking-sensitivity",
        "kind": "sensitivity",
        "methods": [
          "flip2-composition",
          "flip2-train-mean"
        ],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "9a0788c6679363b7e33109d0a4fbf09381aa7d1e5646384286893d02fea61fe3",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "sensitivity",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests.",
          "Sign reversal is diagnostic and never changes the declared score direction."
        ],
        "manifest_sha256": "fc60f1bff46326bd0410063f51c3f4fcec8f0c68a35c8ac11a2938674ef7468e",
        "numerical": {
          "methods": {
            "flip2-composition": {
              "constant_ranking_control": {
                "mse": 286642.9945652174,
                "ndcg": 0.9206667522227658,
                "pearson": null,
                "spearman": null
              },
              "original": {
                "mse": 1042.666651290419,
                "ndcg": 0.954819941821582,
                "pearson": 0.36211321993490453,
                "spearman": 0.41818225155801514
              },
              "scored": 184,
              "sign_reversed_diagnostic": {
                "mse": 1155040.5147361257,
                "ndcg": 0.9152942933597636,
                "pearson": -0.36211321993490453,
                "spearman": -0.41818225155801514
              },
              "unique_predictions": 167
            },
            "flip2-train-mean": {
              "constant_ranking_control": {
                "mse": 286642.9945652174,
                "ndcg": 0.9206667522227658,
                "pearson": null,
                "spearman": null
              },
              "original": {
                "mse": 1056.3941660479575,
                "ndcg": 0.9206667522227658,
                "pearson": null,
                "spearman": null
              },
              "scored": 184,
              "sign_reversed_diagnostic": {
                "mse": 1154852.7513727187,
                "ndcg": 0.9206667522227658,
                "pearson": null,
                "spearman": null
              },
              "unique_predictions": 1
            }
          }
        },
        "operation_id": "agent-1-1-score-and-ranking-sensitivity",
        "table_sha256": "8562d4891270a73026fc7852c05f41206ad52e8fcae1d706d83ec57572c7c949"
      },
      "receipt_sha256": "c12adb8a804547803bed86c06199778fa97809f1b76d5f956eee97682c66e6a7",
      "started_at": "2026-09-23T23:07:43.217472Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:07:44.780917Z",
      "id": "attempt-be0146c9f99fe81246d3e46d",
      "operation": {
        "expected_observation": "Few scored rows or different missing-prediction counts would support a coverage explanation. Complete coverage by both methods would contradict it. Equal incomplete counts alone would not establish that the scored IDs match.",
        "field": null,
        "id": "agent-2-1-whole-population-coverage",
        "kind": "coverage",
        "methods": [
          "flip2-composition",
          "flip2-train-mean"
        ],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "9a0788c6679363b7e33109d0a4fbf09381aa7d1e5646384286893d02fea61fe3",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "coverage",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "fc60f1bff46326bd0410063f51c3f4fcec8f0c68a35c8ac11a2938674ef7468e",
        "numerical": {
          "common_all_methods": 184,
          "coverage": {
            "flip2-composition": {
              "denominator": 184,
              "missing": 0,
              "scored": 184
            },
            "flip2-train-mean": {
              "denominator": 184,
              "missing": 0,
              "scored": 184
            }
          },
          "original_n": 184
        },
        "operation_id": "agent-2-1-whole-population-coverage",
        "table_sha256": "8562d4891270a73026fc7852c05f41206ad52e8fcae1d706d83ec57572c7c949"
      },
      "receipt_sha256": "391889d766d7469ca545307d3958dda01037368d35543a2a8bed6334d7a3fdd2",
      "started_at": "2026-09-23T23:07:43.995477Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:07:45.573920Z",
      "id": "attempt-ee56db39e1c16084cfc0ee64",
      "operation": {
        "expected_observation": "On identical common scored IDs from the preserved snapshot, attenuation or disappearance of the expected composition NDCG advantage of approximately 0.03415 would support coverage contributing to the contrast. Persistence under complete shared coverage would contradict that explanation. If training-mean Spearman remains undefined, its Spearman difference from composition must also remain undefined, never zero.",
        "field": null,
        "id": "agent-2-2-common-row-metric-comparison",
        "kind": "paired",
        "methods": [
          "flip2-composition",
          "flip2-train-mean"
        ],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "9a0788c6679363b7e33109d0a4fbf09381aa7d1e5646384286893d02fea61fe3",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "paired",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "fc60f1bff46326bd0410063f51c3f4fcec8f0c68a35c8ac11a2938674ef7468e",
        "numerical": {
          "pairs": [
            {
              "candidate": "flip2-train-mean",
              "candidate_metrics": {
                "mse": 1056.3941660479575,
                "ndcg": 0.9206667522227658,
                "pearson": null,
                "spearman": null
              },
              "candidate_minus_reference": {
                "mse": 13.727514757538529,
                "ndcg": -0.03415318959881619,
                "pearson": null,
                "spearman": null
              },
              "common_n": 184,
              "original_n": 184,
              "reference": "flip2-composition",
              "reference_metrics": {
                "mse": 1042.666651290419,
                "ndcg": 0.954819941821582,
                "pearson": 0.36211321993490453,
                "spearman": 0.41818225155801514
              }
            }
          ]
        },
        "operation_id": "agent-2-2-common-row-metric-comparison",
        "table_sha256": "8562d4891270a73026fc7852c05f41206ad52e8fcae1d706d83ec57572c7c949"
      },
      "receipt_sha256": "b8b11f0ddae87ff751e58e6a01d983af90fdebcdb52fd2ffa5420a4aedd25e16",
      "started_at": "2026-09-23T23:07:44.781590Z",
      "status": "completed"
    }
  ],
  "catalogue_release_id": "2026-09-20-370b30415b09",
  "claim_level": "exploratory",
  "created_at": "2026-09-23T23:00:33.952383Z",
  "execution": {
    "campaign_id": "campaign-fd29c4413eaf4e24",
    "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
    "codex_calls": 2,
    "reasoning": [
      {
        "cli_version": "codex-cli 0.155.0-alpha.9.2",
        "model": "gpt-6-astra",
        "role": "planner",
        "usage": {
          "cache_write_input_tokens": 0,
          "cached_input_tokens": 0,
          "input_tokens": 12143,
          "output_tokens": 7390,
          "reasoning_output_tokens": 6732
        }
      },
      {
        "cli_version": "codex-cli 0.155.0-alpha.9.2",
        "model": "gpt-6-astra",
        "role": "critic",
        "usage": {
          "cache_write_input_tokens": 0,
          "cached_input_tokens": 0,
          "input_tokens": 16995,
          "output_tokens": 2238,
          "reasoning_output_tokens": 1842
        }
      }
    ]
  },
  "findings": [
    "Artifact verification and all four metric replay checks passed. sdk:train-mean-v1 reproduced the constant control with fitting=train_only. Stale evidence or replay mismatch was not supported.",
    "Both methods scored all 184 rows, with 184 common IDs and no missing predictions. The paired comparison retained the ndcg difference, contradicting the sparse or unequal coverage explanation.",
    "flip2-train-mean produced one unique prediction across 184 rows. Constant ranks have zero variance, making spearman undefined. Its spearman value and paired spearman difference correctly remain null, not zero.",
    "flip2-train-mean and the constant-ranking control both achieved ndcg=0.920667. NDCG normalizes against an ideal ranking; an uninformative predictor need not score zero. High absolute ndcg therefore does not establish ranking discrimination.",
    "flip2-composition achieved ndcg=0.954820 and spearman=0.418182, exceeding the constant control by 0.034153 ndcg. Reversal changed spearman to -0.418182 while ndcg remained 0.915294. NDCG responds to direction here, but remains high across opposite associations."
  ],
  "id": "flip2-discrepancies-campaign-fd29c4413eaf4e24",
  "limitations": [
    "Existing test outcomes are exposed; no independent validation is claimed.",
    "Independent experimental grouping is unavailable; subgroup summaries are descriptive.",
    "Prepared opaque IDs must not be joined to a newly prepared snapshot by ID.",
    "High wavelength is not universally biologically preferable; NDCG is a numerical ranking metric.",
    "Known methodological control, not a novel biological finding.",
    "Exploratory analysis of exposed outcomes; no independent validation or novelty claim.",
    "AI-assisted interpretation requires human scientific review before publication.",
    "All intervals are descriptive and unadjusted; unsuccessful tests remain in the report.",
    "The receipts do not isolate the contributions of gain transformation, tie handling or outcome distribution to the high constant-control ndcg. Specific explanations involving these remain unestablished.",
    "Hash agreement and metric replay establish consistency of the supplied artifacts, without independently establishing implementation correctness.",
    "These are descriptive results on previously exposed public data. They establish no independent validation, discovery, biological mechanism or statistical significance. The independent sampling unit is undocumented, so bootstrap intervals are not justified."
  ],
  "manifest_id": "flip2-discrepancies",
  "outcome": "data_or_method_explanation",
  "plan_sha256": "21b45cf52949e6c2b5ebfd5a03d3c8e8d14fcced8c30987bbbd75d67d4696534",
  "question": "Why can a constant predictor have high NDCG and undefined Spearman correlation?",
  "review": {
    "method": "ai_assisted",
    "status": "pending"
  },
  "schema_version": "1.0",
  "specs": [
    {
      "budget": {
        "campaign_seconds": 3600,
        "codex_calls": 8,
        "codex_seconds": 900,
        "experiment_seconds": 300,
        "followup_rounds": 0,
        "memory_bytes": 2147483648,
        "workspace_bytes": 1073741824
      },
      "catalogue_release_id": "2026-09-20-370b30415b09",
      "created_at": "2026-09-23T23:07:40.719603Z",
      "evidence": [
        {
          "format": "json",
          "id": "flip2-discrepancies-prepared",
          "role": "prepared",
          "semantic_sha256": "f417d43497e0c8c453bd7c66cfea05959b4d6d47462283096e42a2dae8b72419",
          "sha256": "df996aee84bd5ed3e6381f9f6db864a377ab0b4809a9a48ff146d06a3dc069df",
          "uri": null
        },
        {
          "format": "json",
          "id": "flip2-discrepancies-flip2-composition-predictions",
          "role": "flip2-composition-predictions",
          "sha256": "72ea3c506b8645ae6020207f904fbd5c0b09549cd125fcdcabfb487e68681b9c",
          "uri": null
        },
        {
          "format": "json",
          "id": "flip2-discrepancies-flip2-composition-report",
          "role": "flip2-composition-report",
          "sha256": "abd08d3f493428d06f43740c94e268575d755560e72120c73b294127acb395e6",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-composition/report.json"
        },
        {
          "format": "json",
          "id": "flip2-discrepancies-flip2-composition-source",
          "role": "flip2-composition-source",
          "sha256": "7157313eab75ff8f93b934ac186852c8c82f8bb885617cab70c89427c631c1e4",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-composition/source.json"
        },
        {
          "format": "json",
          "id": "flip2-discrepancies-flip2-train-mean-predictions",
          "role": "flip2-train-mean-predictions",
          "sha256": "599dca4e6fa643a7cdfc96ac339033f61e5846ce225af17ac2ac3917ecc14fac",
          "uri": null
        },
        {
          "format": "json",
          "id": "flip2-discrepancies-flip2-train-mean-report",
          "role": "flip2-train-mean-report",
          "sha256": "eb89260f6e138bb9ae2fdf3b882a87fdf8aeabd54a0ac6f17a73f5671c96b4d8",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-train-mean/report.json"
        },
        {
          "format": "json",
          "id": "flip2-discrepancies-flip2-train-mean-source",
          "role": "flip2-train-mean-source",
          "sha256": "8bd712289f136edbfde3fa885166ef85e4f6fe3c4c437810c24dadb87107cc3f",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/flip2-train-mean/source.json"
        },
        {
          "format": "py",
          "id": "flip2-discrepancies-recipe_code",
          "role": "recipe_code",
          "sha256": "b3407940ca4c8a1cdcf3a3bb3849c3e26d7f85272b2ee63087307378ffaa79dc",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/packages/rewirebench/src/rewirebench/adapters/sequence.py"
        },
        {
          "format": "json",
          "id": "flip2-discrepancies-table",
          "role": "table",
          "sha256": "8562d4891270a73026fc7852c05f41206ad52e8fcae1d706d83ec57572c7c949",
          "uri": null
        },
        {
          "format": "json",
          "id": "flip2-discrepancies-receipt",
          "role": "receipt",
          "sha256": "68631e653cd5aa9047b3f4e88cbb3fec94630847c1b73fad8de1018423d81fac",
          "uri": null
        },
        {
          "format": "py",
          "id": "flip2-discrepancies-preparation_code",
          "role": "preparation_code",
          "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
          "uri": null
        },
        {
          "format": "lock",
          "id": "flip2-discrepancies-environment",
          "role": "environment",
          "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
          "uri": null
        }
      ],
      "exposure": {
        "independent_validation": false,
        "previously_exposed": true,
        "usage": "exploration"
      },
      "id": "flip2-discrepancies-round-0",
      "manifest_id": "flip2-discrepancies",
      "manifest_sha256": "fc60f1bff46326bd0410063f51c3f4fcec8f0c68a35c8ac11a2938674ef7468e",
      "permitted_actions": [
        "verify",
        "replay",
        "coverage",
        "paired",
        "subgroups",
        "bootstrap",
        "sensitivity",
        "local_recipe"
      ],
      "plan": {
        "hypotheses": [
          {
            "explanation": "The discrepancy may arise from stale evidence, metric calculation or an inadequate trivial control.",
            "id": "evidence-and-method",
            "tests": [
              {
                "expected_observation": "All available evidence matches its declared byte hash.",
                "field": null,
                "id": "verify-evidence",
                "kind": "verify",
                "methods": [],
                "metric": null,
                "recipe": null
              },
              {
                "expected_observation": "Recorded metrics replay within the stated tolerance.",
                "field": null,
                "id": "replay-metrics",
                "kind": "replay",
                "methods": [],
                "metric": null,
                "recipe": null
              },
              {
                "expected_observation": "The SDK trains only on training labels and reproduces the constant control.",
                "field": null,
                "id": "local-train-mean",
                "kind": "local_recipe",
                "methods": [],
                "metric": null,
                "recipe": "sdk:train-mean-v1"
              }
            ]
          },
          {
            "explanation": "Constant scores would make Spearman undefined. High NDCG may remain attainable without discriminative scores because of the metric's gain and tie conventions; their specific contributions are not established by the supplied expected metrics.",
            "id": "agent-1-metric-degeneracy",
            "tests": [
              {
                "expected_observation": "One unique training-mean score across multiple scored rows, undefined Spearman and unchanged results after score reversal would support score degeneracy. High NDCG for the constant-ranking control would demonstrate that high absolute NDCG requires no discriminative scores here. Multiple unique training-mean scores would contradict strict constancy. Similar composition NDCG despite reversed Spearman would support limited NDCG sensitivity on these outcomes. Reversal remains diagnostic and does not change the declared direction.",
                "field": null,
                "id": "agent-1-1-score-and-ranking-sensitivity",
                "kind": "sensitivity",
                "methods": [
                  "flip2-composition",
                  "flip2-train-mean"
                ],
                "metric": null,
                "recipe": null
              }
            ]
          },
          {
            "explanation": "Sparse or unequal prediction coverage may contribute to undefined correlations or the expected NDCG advantage of composition over the training mean.",
            "id": "agent-2-coverage-selection",
            "tests": [
              {
                "expected_observation": "Few scored rows or different missing-prediction counts would support a coverage explanation. Complete coverage by both methods would contradict it. Equal incomplete counts alone would not establish that the scored IDs match.",
                "field": null,
                "id": "agent-2-1-whole-population-coverage",
                "kind": "coverage",
                "methods": [
                  "flip2-composition",
                  "flip2-train-mean"
                ],
                "metric": null,
                "recipe": null
              },
              {
                "expected_observation": "On identical common scored IDs from the preserved snapshot, attenuation or disappearance of the expected composition NDCG advantage of approximately 0.03415 would support coverage contributing to the contrast. Persistence under complete shared coverage would contradict that explanation. If training-mean Spearman remains undefined, its Spearman difference from composition must also remain undefined, never zero.",
                "field": null,
                "id": "agent-2-2-common-row-metric-comparison",
                "kind": "paired",
                "methods": [
                  "flip2-composition",
                  "flip2-train-mean"
                ],
                "metric": null,
                "recipe": null
              }
            ]
          }
        ],
        "multiple_testing": "descriptive_only",
        "requested_tools": [],
        "stopping_rule": "Interpret results only after the supervisor's verification, metric replay and training-mean control receipts pass; do not repeat those tests. Run coverage before interpreting the other diagnostics. Stop after these three additional tests and retain failed or unsupported explanations. Preserve undefined metrics. No bootstrap is justified because independent_unit is null. Conclusions remain descriptive for exposed public data, without statistical significance, independent validation, discovery or mechanism claims. Request a follow-up only if a new registered test would materially resolve a remaining uncertainty."
      },
      "question": "Why can a constant predictor have high NDCG and undefined Spearman correlation?",
      "round": 0,
      "schema_version": "1.0"
    }
  ],
  "status": "completed",
  "title": "FLIP2 Rhomax: metric interpretation control"
}
