{
  "schemaVersion": "gaiaworld.replogle-k562-result.public.v1",
  "experimentId": "002",
  "benchmarkId": "gaiaworld-replogle-k562-2026-v1",
  "evaluationMode": "retrospective_holdout_gate_0b_hidden_partition",
  "state": "scored",
  "resultClass": "positive_primary",
  "recordedAt": "2026-08-31T02:40:08Z",
  "gate0": {
    "record": "docs/GAIAWORLD_EXPERIMENT_002_GATE0.md",
    "boundaryKind": "0b_declared_hidden_split",
    "honestyNote": "the evaluation partition's outcomes are public since 2022; 'untouched' here means process-hidden under a pre-committed hash rule, strictly weaker than 0a secret outcomes — recorded deliberately"
  },
  "preRevealLock": {
    "lockHash": "367622b3cb7b36abc4a6652919c31b52ccd193628e4474092c166777af6206d5",
    "lockedAt": "2026-08-31T02:31:21.796Z",
    "partitionManifestSha256": "055e239bb6136597059af110f07f7a16dba2d9b25729cde97808ada103264905",
    "panelSha256": "79cc63cb3f34a582f4f36bab81170fe499ff0c5c8b819a2498506d64dd321f4d",
    "modelSha256": "07ade47332e3636b0bd7e862687a18986c92696b9d47f14c6967b6e8aaad76aa",
    "acquisitionSha256": "85b3af32f26e18da87dac2685d793327a6d71d3aad743e82eb20d7a748d37433"
  },
  "scoreArtifact": {
    "scoreHash": "483c1b4b3b2342c3b43cc15011eac0678c77d229c5b0509fb89ac800d96966c2",
    "fileSha256": "65ebcf589d6e0bc36cc20705cd804d0751edc1fddd04bb0d18905d7d71bf68b9",
    "path": "data/benchmarks/gaiaworld-k562-score-artifact.json"
  },
  "denominator": {
    "cases": 200,
    "dimensions": 100,
    "values": 20000,
    "unit": "log1p_cp10k_delta"
  },
  "metricContract": {
    "primary": "panelMAE",
    "lowerIsBetter": true,
    "secondary": ["panelRMSE", "caseWinRate", "panelPDSL1MidrankNoExclusion"],
    "panelPDS": {
      "distance": "l1",
      "tiePolicy": "midrank",
      "targetGeneExclusion": false,
      "featureSpace": "pre-frozen-100-gene-panel",
      "officialArcLeaderboardMetric": false
    },
    "officialArcLeaderboardMetrics": false
  },
  "lanes": [
    {
      "name": "no-change",
      "modelId": "replogle-k562-no-change-v1",
      "panelMAE": 0.047050581038723414,
      "panelRMSE": 0.06134543034042813,
      "panelPDSL1MidrankNoExclusion": 0.5024999999999998
    },
    {
      "name": "empirical-mean",
      "modelId": "replogle-k562-empirical-mean-v1",
      "panelMAE": 0.02523383613556392,
      "panelRMSE": 0.04183061373341174,
      "panelPDSL1MidrankNoExclusion": 0.5024999999999998
    },
    {
      "name": "baseline-residual-ridge",
      "modelId": "replogle-k562-baseline-residual-ridge-v1",
      "panelMAE": 0.024975421304019876,
      "panelRMSE": 0.0416362216108318,
      "panelPDSL1MidrankNoExclusion": 0.51225,
      "caseWinRateVersusNoChange": 0.99,
      "caseWinRateVersusEmpiricalMean": 0.175,
      "relativePanelMAEImprovementVersusNoChange": 0.46920969780645766,
      "relativePanelMAEImprovementVersusEmpiricalMean": 0.01024075418559993,
      "panelPDSImprovementVersusBaselines": 0.009750000000000009,
      "abstentions": 0,
      "baselineStar": "empirical-mean",
      "alpha": 1,
      "gamma": 1,
      "tau": 0
    }
  ],
  "interpretation": {
    "primary": "positive",
    "primaryStatement": "The frozen baseline-residual-ridge model scored panel MAE 0.0249754 on the untouched 200-case K562 partition, strictly lower than both declared baselines (no-change 0.0470506, empirical-mean 0.0252338), meeting the pre-written success criterion.",
    "marginStatement": "The positive margin over the stronger baseline is narrow: 0.0002584 absolute panel MAE (~1.02% relative). Against the empirical-mean baseline the model won 35 cases, lost 9, and tied 156; the top 5 winning cases contribute ~47% of the total gain. The effect is concentrated in the ~quarter of cases whose perturbation target is present in the 2058-gene measurement matrix; the remaining cases fell back to baseline exactly (tau = 0 left the explicit abstention gate open; zero-residual fallback handled unsupported cases).",
    "secondary": "supporting",
    "secondaryStatement": "The model's panel discrimination diagnostic (0.51225) exceeded both baselines (0.5025) by +0.0098. This diagnostic is secondary, non-overriding, and not an official Arc leaderboard metric.",
    "developmentConsistency": "the narrow positive margin is consistent with the development-pool evidence (out-of-fold development panel-MAE 0.02401 for the model vs 0.02428 for the baseline, ~1.1%)"
  },
  "scientificBoundary": "Retrospective held-out 100-gene panel result on a Gate 0b hidden partition of a public dataset (Replogle et al. 2022, PMID 35688146, CC-BY-4.0). One narrow positive primary result on one benchmark does not establish general biological prediction ability, causal validity, patient benefit, safety, efficacy, clinical utility, or any official leaderboard standing. Benchmark performance is research evidence about a model and an evaluation contract."
}
