{
  "artifacts": [
    {
      "format": "json",
      "id": "proteingym-amfr-discrepancies-prepared",
      "role": "prepared",
      "semantic_sha256": "9f0e0cce34942797364990ef4dc9fb8b382f7b2fd59e6c72de23efc2992fb148",
      "sha256": "d9e220a0e2b0675e73f4b79b4ae45b8f807dedbb886bbd4b2f8e886822ddf57f",
      "uri": null
    },
    {
      "format": "json",
      "id": "proteingym-amfr-discrepancies-proteingym-esm2-predictions",
      "role": "proteingym-esm2-predictions",
      "sha256": "7ab0c8901406d5c1ab759038c8beef235721938394cfdf192c46671e98e0bbca",
      "uri": null
    },
    {
      "format": "json",
      "id": "proteingym-amfr-discrepancies-proteingym-esm2-report",
      "role": "proteingym-esm2-report",
      "sha256": "6a14b3866141d5779c76da8071d85c73b7cce0dc4e101c93122706873e847d26",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/proteingym-esm2/report.json"
    },
    {
      "format": "json",
      "id": "proteingym-amfr-discrepancies-proteingym-esm2-source",
      "role": "proteingym-esm2-source",
      "sha256": "90888f1f18625cdc2e9fb0bf7ecc1b1c4c2d6b700e8b4971e4349297926f996f",
      "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/proteingym-esm2/retrieval.json"
    },
    {
      "format": "json",
      "id": "proteingym-amfr-discrepancies-table",
      "role": "table",
      "sha256": "4fac11aacc0f00a70e385ed7b8d73427aaa0716906a74e93bc3d48df5ff43f9f",
      "uri": null
    },
    {
      "format": "json",
      "id": "proteingym-amfr-discrepancies-receipt",
      "role": "receipt",
      "sha256": "9f117c65a3c860b83735d4874d06a807ee1a1523464e9afca493e7dcd9e59b56",
      "uri": null
    },
    {
      "format": "py",
      "id": "proteingym-amfr-discrepancies-preparation_code",
      "role": "preparation_code",
      "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
      "uri": null
    },
    {
      "format": "lock",
      "id": "proteingym-amfr-discrepancies-environment",
      "role": "environment",
      "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
      "uri": null
    }
  ],
  "attempts": [
    {
      "error": null,
      "finished_at": "2026-09-23T23:02:26.793171Z",
      "id": "attempt-22ea1b92829cf8f305e0bedf",
      "operation": {
        "expected_observation": "All available evidence matches its declared byte hash.",
        "field": null,
        "id": "verify-evidence",
        "kind": "verify",
        "methods": [],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "8db3245b67bd29c13783e6dae9e025dd600f7eaae1458c3b3ff95af48743de4f",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "verify",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "dc4a5a77d78eedc53eb0f55632fe0f653965fc8e8a90675d3640bd9a11a77523",
        "numerical": {
          "checks": [
            {
              "artifact_id": "proteingym-amfr-discrepancies-prepared",
              "status": "passed"
            },
            {
              "artifact_id": "proteingym-amfr-discrepancies-proteingym-esm2-predictions",
              "status": "passed"
            },
            {
              "artifact_id": "proteingym-amfr-discrepancies-proteingym-esm2-report",
              "status": "passed"
            },
            {
              "artifact_id": "proteingym-amfr-discrepancies-proteingym-esm2-source",
              "status": "passed"
            },
            {
              "artifact_id": "proteingym-amfr-discrepancies-table",
              "status": "passed"
            },
            {
              "artifact_id": "proteingym-amfr-discrepancies-receipt",
              "status": "passed"
            },
            {
              "artifact_id": "proteingym-amfr-discrepancies-preparation_code",
              "status": "passed"
            },
            {
              "artifact_id": "proteingym-amfr-discrepancies-environment",
              "status": "passed"
            }
          ],
          "identifiers_unique": true,
          "methods": [
            "proteingym-esm2"
          ],
          "rows": 2972
        },
        "operation_id": "verify-evidence",
        "table_sha256": "4fac11aacc0f00a70e385ed7b8d73427aaa0716906a74e93bc3d48df5ff43f9f"
      },
      "receipt_sha256": "7413427155f579bc581ca373f0eccd423dc4822314e3f584c131a7c2a9766955",
      "started_at": "2026-09-23T23:02:25.660627Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:02:27.696298Z",
      "id": "attempt-36baf8ae9cf4647cf7ca6f0b",
      "operation": {
        "expected_observation": "Recorded metrics replay within the stated tolerance.",
        "field": null,
        "id": "replay-metrics",
        "kind": "replay",
        "methods": [],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "8db3245b67bd29c13783e6dae9e025dd600f7eaae1458c3b3ff95af48743de4f",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "replay",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "dc4a5a77d78eedc53eb0f55632fe0f653965fc8e8a90675d3640bd9a11a77523",
        "numerical": {
          "checks": [
            {
              "actual": 0.394,
              "expected": 0.394,
              "method": "proteingym-esm2",
              "metric": "AUC",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": -0.139,
              "expected": -0.139,
              "method": "proteingym-esm2",
              "metric": "MCC",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.44,
              "expected": 0.44,
              "method": "proteingym-esm2",
              "metric": "NDCG",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": -0.209,
              "expected": -0.209,
              "method": "proteingym-esm2",
              "metric": "Spearman",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            },
            {
              "actual": 0.057,
              "expected": 0.057,
              "method": "proteingym-esm2",
              "metric": "Top_recall",
              "metric_available": true,
              "status": "passed",
              "tolerance": 1e-09
            }
          ],
          "coverage": {
            "proteingym-esm2": {
              "denominator": 2972,
              "missing": 0,
              "scored": 2972
            }
          },
          "metrics": {
            "proteingym-esm2": {
              "AUC": 0.394,
              "MCC": -0.139,
              "NDCG": 0.44,
              "Spearman": -0.209,
              "Top_recall": 0.057
            }
          }
        },
        "operation_id": "replay-metrics",
        "table_sha256": "4fac11aacc0f00a70e385ed7b8d73427aaa0716906a74e93bc3d48df5ff43f9f"
      },
      "receipt_sha256": "192a660bf3706d64a2c40a545970d4799d0d68d511c6f6eb9ccb4a53ad60dd6c",
      "started_at": "2026-09-23T23:02:26.793846Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:02:28.606246Z",
      "id": "attempt-7074d58b8d58e1a4c1c6dea2",
      "operation": {
        "expected_observation": "Missing predictions support a possible selection concern, but do not establish its effect on association. Complete coverage rules out missing-prediction exclusion within the prepared population. The mutation-class test below supplies category-specific coverage.",
        "field": null,
        "id": "agent-1-1-whole-population-coverage",
        "kind": "coverage",
        "methods": [
          "proteingym-esm2"
        ],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "8db3245b67bd29c13783e6dae9e025dd600f7eaae1458c3b3ff95af48743de4f",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "coverage",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests."
        ],
        "manifest_sha256": "dc4a5a77d78eedc53eb0f55632fe0f653965fc8e8a90675d3640bd9a11a77523",
        "numerical": {
          "common_all_methods": 2972,
          "coverage": {
            "proteingym-esm2": {
              "denominator": 2972,
              "missing": 0,
              "scored": 2972
            }
          },
          "original_n": 2972
        },
        "operation_id": "agent-1-1-whole-population-coverage",
        "table_sha256": "4fac11aacc0f00a70e385ed7b8d73427aaa0716906a74e93bc3d48df5ff43f9f"
      },
      "receipt_sha256": "b6aca0198c8c8ab394cafac1be598088cc23476a9fe5366c0984dc22f2636904",
      "started_at": "2026-09-23T23:02:27.696970Z",
      "status": "completed"
    },
    {
      "error": null,
      "finished_at": "2026-09-23T23:02:29.493438Z",
      "id": "attempt-08f348d8667e33bbe9b880d8",
      "operation": {
        "expected_observation": "Few unique scores support a resolution concern; many distinct scores weaken it. Compare constant-ranking controls only where metrics are defined, preserving undefined results. Sign reversal predictably reverses Spearman's sign; improvement alone cannot establish an orientation error or justify changing the declared direction.",
        "field": null,
        "id": "agent-2-1-score-resolution-and-controls",
        "kind": "sensitivity",
        "methods": [
          "proteingym-esm2"
        ],
        "metric": null,
        "recipe": null
      },
      "plan_sha256": "8db3245b67bd29c13783e6dae9e025dd600f7eaae1458c3b3ff95af48743de4f",
      "receipt": {
        "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
        "kind": "sensitivity",
        "limitations": [
          "Previously explored benchmark outcomes; this is not independent validation.",
          "Descriptive analysis only; multiple comparisons are not confirmatory tests.",
          "Sign reversal is diagnostic and never changes the declared score direction."
        ],
        "manifest_sha256": "dc4a5a77d78eedc53eb0f55632fe0f653965fc8e8a90675d3640bd9a11a77523",
        "numerical": {
          "methods": {
            "proteingym-esm2": {
              "constant_ranking_control": {
                "AUC": 0.5,
                "MCC": 0,
                "NDCG": 0.393,
                "Spearman": null,
                "Top_recall": 1
              },
              "original": {
                "AUC": 0.394,
                "MCC": -0.139,
                "NDCG": 0.44,
                "Spearman": -0.209,
                "Top_recall": 0.057
              },
              "scored": 2972,
              "sign_reversed_diagnostic": {
                "AUC": 0.606,
                "MCC": 0.139,
                "NDCG": 0.71,
                "Spearman": 0.209,
                "Top_recall": 0.232
              },
              "unique_predictions": 2970
            }
          }
        },
        "operation_id": "agent-2-1-score-resolution-and-controls",
        "table_sha256": "4fac11aacc0f00a70e385ed7b8d73427aaa0716906a74e93bc3d48df5ff43f9f"
      },
      "receipt_sha256": "25807e2d8971d89f7da1b1b075454e3373c83b6aaf5d4b5bbfb7cf9b17b8ea69",
      "started_at": "2026-09-23T23:02:28.607092Z",
      "status": "completed"
    }
  ],
  "catalogue_release_id": "2026-09-20-370b30415b09",
  "claim_level": "exploratory",
  "created_at": "2026-09-23T23:00:33.952383Z",
  "execution": {
    "campaign_id": "campaign-fd29c4413eaf4e24",
    "code_sha256": "5c41ffd01ea787711cf00a8bc493aafc70e28b797b3a273da319b4954026c596",
    "codex_calls": 2,
    "reasoning": [
      {
        "cli_version": "codex-cli 0.155.0-alpha.9.2",
        "model": "gpt-6-astra",
        "role": "planner",
        "usage": {
          "cache_write_input_tokens": 0,
          "cached_input_tokens": 0,
          "input_tokens": 11423,
          "output_tokens": 3595,
          "reasoning_output_tokens": 3106
        }
      },
      {
        "cli_version": "codex-cli 0.155.0-alpha.9.2",
        "model": "gpt-6-astra",
        "role": "critic",
        "usage": {
          "cache_write_input_tokens": 0,
          "cached_input_tokens": 0,
          "input_tokens": 14356,
          "output_tokens": 2714,
          "reasoning_output_tokens": 2315
        }
      }
    ]
  },
  "findings": [
    "The single-versus-multiple substitution question remains unresolved: no mutation_class subgroup receipt supplies category counts, coverage or Spearman coefficients. Pooled Spearman=-0.209 cannot establish either within-class association.",
    "Eight artifact checks passed, identifiers were unique, and all five reported metrics replayed within 1e-9. This supports internal consistency, not independent validation or correct score semantics.",
    "proteingym-esm2 scored 2972/2972 prepared rows, ruling out missing-prediction exclusion within this population. Its 2970 unique predictions weaken the limited-score-resolution explanation.",
    "Sign reversal produced Spearman=0.209 and AUC=0.606, versus -0.209 and 0.394 originally. These expected transformations do not establish an orientation error or justify reversing the declared direction.",
    "The constant-ranking control returned Spearman=null and Top_recall=1, versus original Top_recall=0.057. Preserve the undefined correlation; the recall result requires clarification of metric and tie handling before interpreting it as ranking performance."
  ],
  "id": "proteingym-amfr-discrepancies-campaign-fd29c4413eaf4e24",
  "limitations": [
    "Existing test outcomes are exposed; no independent validation is claimed.",
    "Independent experimental grouping is unavailable; subgroup summaries are descriptive.",
    "Prepared opaque IDs must not be joined to a newly prepared snapshot by ID.",
    "One protein construct, complete selected assay but partial benchmark suite.",
    "Official three-decimal metrics and score sign are preserved; no post-hoc sign reversal.",
    "Upstream archive checksums were observed locally, not independently source-verified.",
    "Exploratory analysis of exposed outcomes; no independent validation or novelty claim.",
    "AI-assisted interpretation requires human scientific review before publication.",
    "All intervals are descriptive and unadjusted; unsuccessful tests remain in the report.",
    "Requested tool without a supplied execution receipt: subgroups(methods=['proteingym-esm2'], metric='Spearman', field='mutation_class', recipe=null). Different defined coefficients would support descriptive class heterogeneity; similar negative coefficients would weaken it. An absent category or undefined coefficient would leave the comparison unresolved.",
    "Complete prepared-population coverage does not establish upstream representativeness. Replay of reported three-decimal metrics does not establish unrounded accuracy.",
    "Previously exposed data and undocumented independent sampling units permit descriptive conclusions only; bootstrap intervals, significance claims and biological mechanisms are unsupported.",
    "No follow-up rounds remain."
  ],
  "manifest_id": "proteingym-amfr-discrepancies",
  "outcome": "inconclusive",
  "plan_sha256": "61a9dba6fd12fe503b3b869444fa42f6a32b7d315269d38dbea77dea44f423ff",
  "question": "Does the negative association differ between single and multiple substitutions?",
  "review": {
    "method": "ai_assisted",
    "status": "pending"
  },
  "schema_version": "1.0",
  "specs": [
    {
      "budget": {
        "campaign_seconds": 3600,
        "codex_calls": 8,
        "codex_seconds": 900,
        "experiment_seconds": 300,
        "followup_rounds": 0,
        "memory_bytes": 2147483648,
        "workspace_bytes": 1073741824
      },
      "catalogue_release_id": "2026-09-20-370b30415b09",
      "created_at": "2026-09-23T23:02:25.658027Z",
      "evidence": [
        {
          "format": "json",
          "id": "proteingym-amfr-discrepancies-prepared",
          "role": "prepared",
          "semantic_sha256": "9f0e0cce34942797364990ef4dc9fb8b382f7b2fd59e6c72de23efc2992fb148",
          "sha256": "d9e220a0e2b0675e73f4b79b4ae45b8f807dedbb886bbd4b2f8e886822ddf57f",
          "uri": null
        },
        {
          "format": "json",
          "id": "proteingym-amfr-discrepancies-proteingym-esm2-predictions",
          "role": "proteingym-esm2-predictions",
          "sha256": "7ab0c8901406d5c1ab759038c8beef235721938394cfdf192c46671e98e0bbca",
          "uri": null
        },
        {
          "format": "json",
          "id": "proteingym-amfr-discrepancies-proteingym-esm2-report",
          "role": "proteingym-esm2-report",
          "sha256": "6a14b3866141d5779c76da8071d85c73b7cce0dc4e101c93122706873e847d26",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/proteingym-esm2/report.json"
        },
        {
          "format": "json",
          "id": "proteingym-amfr-discrepancies-proteingym-esm2-source",
          "role": "proteingym-esm2-source",
          "sha256": "90888f1f18625cdc2e9fb0bf7ecc1b1c4c2d6b700e8b4971e4349297926f996f",
          "uri": "https://raw.githubusercontent.com/rewire-bio/rewire-benchmarks/ca73fa47136d182f2d4ddb083d084712198fc0e2/research/local-runs-2026-09-20/proteingym-esm2/retrieval.json"
        },
        {
          "format": "json",
          "id": "proteingym-amfr-discrepancies-table",
          "role": "table",
          "sha256": "4fac11aacc0f00a70e385ed7b8d73427aaa0716906a74e93bc3d48df5ff43f9f",
          "uri": null
        },
        {
          "format": "json",
          "id": "proteingym-amfr-discrepancies-receipt",
          "role": "receipt",
          "sha256": "9f117c65a3c860b83735d4874d06a807ee1a1523464e9afca493e7dcd9e59b56",
          "uri": null
        },
        {
          "format": "py",
          "id": "proteingym-amfr-discrepancies-preparation_code",
          "role": "preparation_code",
          "sha256": "8a1d691de8fdd9fcf8242fe6fa2cfe2703a1c3281b880fe2a226cb8bd946fd1b",
          "uri": null
        },
        {
          "format": "lock",
          "id": "proteingym-amfr-discrepancies-environment",
          "role": "environment",
          "sha256": "748257abb3664595db1ebf892799e91c6f1a4ca79b0e65eccb3958cf5644a671",
          "uri": null
        }
      ],
      "exposure": {
        "independent_validation": false,
        "previously_exposed": true,
        "usage": "exploration"
      },
      "id": "proteingym-amfr-discrepancies-round-0",
      "manifest_id": "proteingym-amfr-discrepancies",
      "manifest_sha256": "dc4a5a77d78eedc53eb0f55632fe0f653965fc8e8a90675d3640bd9a11a77523",
      "permitted_actions": [
        "verify",
        "replay",
        "coverage",
        "paired",
        "subgroups",
        "bootstrap",
        "sensitivity",
        "local_recipe"
      ],
      "plan": {
        "hypotheses": [
          {
            "explanation": "The discrepancy may arise from stale evidence, metric calculation or an inadequate trivial control.",
            "id": "evidence-and-method",
            "tests": [
              {
                "expected_observation": "All available evidence matches its declared byte hash.",
                "field": null,
                "id": "verify-evidence",
                "kind": "verify",
                "methods": [],
                "metric": null,
                "recipe": null
              },
              {
                "expected_observation": "Recorded metrics replay within the stated tolerance.",
                "field": null,
                "id": "replay-metrics",
                "kind": "replay",
                "methods": [],
                "metric": null,
                "recipe": null
              }
            ]
          },
          {
            "explanation": "Missing predictions could restrict which variants contribute to the reported association.",
            "id": "agent-1-coverage-selection",
            "tests": [
              {
                "expected_observation": "Missing predictions support a possible selection concern, but do not establish its effect on association. Complete coverage rules out missing-prediction exclusion within the prepared population. The mutation-class test below supplies category-specific coverage.",
                "field": null,
                "id": "agent-1-1-whole-population-coverage",
                "kind": "coverage",
                "methods": [
                  "proteingym-esm2"
                ],
                "metric": null,
                "recipe": null
              }
            ]
          },
          {
            "explanation": "Limited score diversity could reduce the information supporting the reported ranking metrics.",
            "id": "agent-2-limited-score-resolution",
            "tests": [
              {
                "expected_observation": "Few unique scores support a resolution concern; many distinct scores weaken it. Compare constant-ranking controls only where metrics are defined, preserving undefined results. Sign reversal predictably reverses Spearman's sign; improvement alone cannot establish an orientation error or justify changing the declared direction.",
                "field": null,
                "id": "agent-2-1-score-resolution-and-controls",
                "kind": "sensitivity",
                "methods": [
                  "proteingym-esm2"
                ],
                "metric": null,
                "recipe": null
              }
            ]
          }
        ],
        "multiple_testing": "descriptive_only",
        "requested_tools": [
          "subgroups(methods=['proteingym-esm2'], metric='Spearman', field='mutation_class', recipe=null): compare predeclared single- and multiple-substitution categories, reporting counts, scored coverage and signed coefficients. A negative coefficient confined to one class supports class specificity; similarly negative coefficients weaken it. Preserve undefined estimates and report magnitude differences descriptively."
        ],
        "stopping_rule": "Interpret additional results only after supervisor verification and replay succeed. Run each test once and retain unsupported explanations. These tests assess coverage and score resolution but cannot answer the single-versus-multiple question without the registered mutation_class subgroup test. Independence is undocumented, so do not bootstrap or claim significance, mechanism, discovery or independent validation. Request follow-up only when a new registered test materially resolves remaining uncertainty."
      },
      "question": "Does the negative association differ between single and multiple substitutions?",
      "round": 0,
      "schema_version": "1.0"
    }
  ],
  "status": "completed",
  "title": "ProteinGym AMFR: negative correlation"
}
