{
  "schemaVersion": "gaiaworld.arc-h1-panel-result.public.v1",
  "experimentId": "001",
  "benchmarkId": "gaiaworld-arc-h1-2025-v1",
  "evaluationMode": "retrospective_holdout",
  "state": "scored",
  "resultClass": "negative_primary_with_secondary_discrimination_signal",
  "recordedAt": "2026-08-30T14:48:24Z",
  "preRevealLock": {
    "lockHash": "07a540e1e60e130547e434ff6300b030768a265ac02edca7535b724ce5a07108",
    "receiptFileSha256": "dc98b416ff371f77339e3cde96632d788e39b534f8117037437e6c8070bb76cf",
    "auditRecordedAt": "2026-08-30T04:20:52Z"
  },
  "scoreArtifact": {
    "scoreHash": "d896bc70b7555a867304b90a6237ade5025176fe556abbfef318e8d966fe00da",
    "fileSha256": "f38b1f9ff029991cf140422b67f05e70da053cb87be7c62517e34c51fa3e9ecf",
    "roundTripSha256": "f38b1f9ff029991cf140422b67f05e70da053cb87be7c62517e34c51fa3e9ecf",
    "gcsPrefix": "gs://gaialab-506816-gaiaworld-001-artifacts/scores/arc-h1/d896bc70b7555a867304b90a6237ade5025176fe556abbfef318e8d966fe00da/"
  },
  "denominator": {
    "cases": 100,
    "dimensions": 100,
    "values": 10000,
    "unit": "log1p_cp10k_delta"
  },
  "metricContract": {
    "primary": "panelMAE",
    "lowerIsBetter": true,
    "secondary": [
      "panelRMSE",
      "caseWinRate",
      "panelPDSL1MidrankNoExclusion"
    ],
    "panelPDS": {
      "distance": "l1",
      "tiePolicy": "midrank",
      "targetGeneExclusion": false,
      "featureSpace": "pre-frozen-100-gene-panel",
      "officialArcLeaderboardMetric": false
    },
    "officialArcLeaderboardMetrics": false
  },
  "lanes": [
    {
      "name": "no-change",
      "modelId": "arc-h1-no-change-v1",
      "manifestHash": "8aef3201ffdb5ed7bd2370bef04fd788b626d023e3a4a744610e9ae3ce76fb04",
      "predictionsHash": "5c086e12430dfeee75f4d2f5033393d229d1a3ded6a4b0ee92bbef7cd8d5556c",
      "panelMAE": 0.07615881768500006,
      "panelRMSE": 0.14742479246557294,
      "panelPDSL1MidrankNoExclusion": 0.5050000000000001
    },
    {
      "name": "empirical-mean",
      "modelId": "arc-h1-empirical-mean-v1",
      "manifestHash": "4065fd7e09a35b2d21173776dc1e1a38a36ccfc25f6b8e7fba3b2a8e36bbb168",
      "predictionsHash": "4981237202859fe17f7d4bd3e8ebcb989b2287d8a8ebaee65b5d4d12ee148a32",
      "panelMAE": 0.08194923586099996,
      "panelRMSE": 0.14474121927189892,
      "panelPDSL1MidrankNoExclusion": 0.5049999999999999,
      "caseWinRateVersusNoChange": 0.4
    },
    {
      "name": "target-ridge",
      "modelId": "arc-h1-target-coexpression-ridge-v1",
      "manifestHash": "412c0f483902204e327af2ee6247784f7889a43954be00e6529ed4e80a8b66c1",
      "predictionsHash": "a5fae28c010e08fe5109fc2940ea46f5139864c87c1aefc055fef9bb8d387e14",
      "panelMAE": 0.13257230776900006,
      "panelRMSE": 0.20639483736455136,
      "panelPDSL1MidrankNoExclusion": 0.5332000000000001,
      "caseWinRateVersusNoChange": 0.11,
      "caseWinRateVersusEmpiricalMean": 0.08,
      "relativePanelMAEImprovementVersusNoChange": -0.7407348459285624,
      "relativePanelMAEImprovementVersusEmpiricalMean": -0.6177369608896117,
      "panelPDSImprovementVersusNoChange": 0.028200000000000003,
      "panelPDSImprovementVersusEmpiricalMean": 0.028200000000000225
    }
  ],
  "directEvidenceLane": {
    "admittedEffects": 0,
    "abstentions": 100,
    "coverage": 0,
    "interpretation": "No Arc-compatible direct transition effects were admitted. This lane remains 100/100 abstentions and was not converted into a numeric prediction."
  },
  "interpretation": {
    "primary": "negative",
    "primaryStatement": "The frozen target-ridge model did not improve held-out perturbation-magnitude prediction over either declared baseline on the pre-frozen 100-gene panel.",
    "secondary": "mixed",
    "secondaryStatement": "The target-ridge model showed a modest +0.0282 improvement over both baselines in the predeclared Arc-inspired panel discrimination diagnostic, but that diagnostic is secondary and is not official Arc PDS.",
    "nextExperimentConstraint": "Experiment #001 is revealed and closed to tuning. Any model revision informed by this result must be evaluated as a new experiment on a genuinely new untouched holdout."
  },
  "scientificBoundary": "This retrospective 100-gene panel result does not establish causal validity, general biological prediction ability, patient benefit, safety, efficacy, clinical utility, or an official Arc leaderboard result. DES and official full-transcriptome Arc metrics were not computed."
}
