{
  "diagnostic_schema": "nmd-vcell-vcc-metric-diagnostic/1.0",
  "feature_build": "VCC-METRICS-20260729-02",
  "assessed_at": "2026-07-29",
  "execution_state": "EXECUTED_AGGREGATE_DELTA_DIAGNOSTIC_NOT_OFFICIAL_VCC",
  "scientific_context": {
    "source_stage": "4.16i",
    "source_mode": "real",
    "source_result_label": "NO_DIRECTIONAL_TRANSFER_SUPPORT",
    "post_pilot_diagnostic_only": true,
    "training_context": "HepG2 aggregate perturbation deltas",
    "external_truth_context": "Frangieh external perturbation response panel",
    "n_external_targets": 55,
    "n_response_coordinates": 1907,
    "prediction_unit": "target_level_aggregate_delta"
  },
  "formula_alignment": {
    "mae_delta": "Mean absolute difference between predicted and observed aggregate perturbation deltas, averaged within target then across targets.",
    "delta_pearson": "Pearson correlation between predicted and observed perturbation deltas within each target.",
    "pds_l1_raw": "cell-eval-aligned L1 rank retrieval on aggregate deltas; the perturbed target gene is excluded when present.",
    "pds_norm_matched": "Prediction vectors are scaled to paired observed L2 norms before L1 retrieval; uses hidden truth and is sensitivity-only.",
    "upstream": "https://github.com/ArcInstitute/cell-eval/blob/main/src/cell_eval/metrics/_anndata.py",
    "adapter_difference": "The upstream package consumes predicted and observed AnnData and can compute cell-level and DE metrics. This adapter consumes frozen target-level aggregate deltas only."
  },
  "uncertainty": {
    "unit": "external_perturbation_target",
    "method": "nonparametric target bootstrap",
    "seed": 20260729,
    "replicates": 10000,
    "note": "Intervals quantify target-sampling variation within this diagnostic panel, not donor, dataset or disease-stage uncertainty."
  },
  "scoreboard": [
    {
      "method_id": "safe_ridge_transfer",
      "label": "Safe ridge transfer",
      "role": "evaluated_model",
      "metrics": {
        "rmse_delta": {
          "estimate": 1.092833214922748,
          "median": 0.9014754826430739,
          "ci95": [
            0.9236388997851348,
            1.2967369277787621
          ],
          "n_targets": 55,
          "direction": "lower_is_better"
        },
        "mae_delta": {
          "estimate": 0.18614194742141588,
          "median": 0.16598531848327125,
          "ci95": [
            0.16646300438486353,
            0.20900720836188075
          ],
          "n_targets": 55,
          "direction": "lower_is_better"
        },
        "delta_pearson": {
          "estimate": -0.01483165414561016,
          "median": -0.002507227576440636,
          "ci95": [
            -0.04278290905927561,
            0.013204431959844807
          ],
          "n_targets": 55,
          "direction": "higher_is_better"
        },
        "delta_spearman": {
          "estimate": -0.0013054617186833357,
          "median": 0.02020309801511988,
          "ci95": [
            -0.04189782076147989,
            0.0404580773372938
          ],
          "n_targets": 55,
          "direction": "higher_is_better"
        },
        "raw_cosine": {
          "estimate": -0.014369727127182623,
          "median": -0.0032922388558917527,
          "ci95": [
            -0.04188232406264992,
            0.013029767583991107
          ],
          "n_targets": 55,
          "direction": "higher_is_better"
        },
        "residual_cosine_vs_train_mean": {
          "estimate": -0.025803286043391577,
          "median": -0.019486474006846694,
          "ci95": [
            -0.04945873070342422,
            -0.0020961425871275346
          ],
          "n_targets": 55,
          "direction": "higher_is_better"
        },
        "pds_l1_raw": {
          "estimate": 0.5147107438016529,
          "median": 0.509090909090909,
          "ci95": [
            0.4376859504132231,
            0.5917438016528924
          ],
          "n_targets": 55,
          "direction": "higher_is_better"
        },
        "effect_size_spearman": {
          "estimate": -0.05541125541125541,
          "median": null,
          "ci95": [
            null,
            null
          ],
          "n_targets": 55,
          "direction": "higher_is_better",
          "definition": "Spearman correlation across perturbations between predicted and observed delta L2 norms."
        }
      },
      "interpretation": "Post-pilot diagnostic model; source conclusion remains NO_DIRECTIONAL_TRANSFER_SUPPORT."
    },
    {
      "method_id": "training_response_mean",
      "label": "Training-response mean",
      "role": "baseline",
      "metrics": {
        "rmse_delta": {
          "estimate": 1.0925456725332876,
          "median": 0.9012560486840381,
          "ci95": [
            0.9234049209592016,
            1.2967007049818533
          ],
          "n_targets": 55,
          "direction": "lower_is_better"
        },
        "mae_delta": {
          "estimate": 0.18444890067018443,
          "median": 0.16265647094626634,
          "ci95": [
            0.16472037398487277,
            0.20736450608705262
          ],
          "n_targets": 55,
          "direction": "lower_is_better"
        },
        "delta_pearson": {
          "estimate": -0.016084257247051453,
          "median": -0.005860289673441154,
          "ci95": [
            -0.04449304748575291,
            0.012588747829121104
          ],
          "n_targets": 55,
          "direction": "higher_is_better"
        },
        "delta_spearman": {
          "estimate": -0.002686417538963046,
          "median": 0.0011989865529231892,
          "ci95": [
            -0.043748282354426384,
            0.039411723331496805
          ],
          "n_targets": 55,
          "direction": "higher_is_better"
        },
        "raw_cosine": {
          "estimate": -0.016005371414858378,
          "median": -0.005723953744722612,
          "ci95": [
            -0.04414877285556557,
            0.011966136389764525
          ],
          "n_targets": 55,
          "direction": "higher_is_better"
        },
        "residual_cosine_vs_train_mean": {
          "estimate": null,
          "median": null,
          "ci95": [
            null,
            null
          ],
          "n_targets": 0,
          "direction": "higher_is_better"
        },
        "pds_l1_raw": {
          "estimate": 0.5133884297520661,
          "median": 0.509090909090909,
          "ci95": [
            0.4366942148760331,
            0.5900909090909089
          ],
          "n_targets": 55,
          "direction": "higher_is_better"
        },
        "effect_size_spearman": {
          "estimate": null,
          "median": null,
          "ci95": [
            null,
            null
          ],
          "n_targets": 55,
          "direction": "higher_is_better",
          "definition": "Spearman correlation across perturbations between predicted and observed delta L2 norms."
        }
      },
      "interpretation": "Comparator evaluated on the identical 55-target aggregate-delta matrix."
    },
    {
      "method_id": "zero_change",
      "label": "Zero change",
      "role": "baseline",
      "metrics": {
        "rmse_delta": {
          "estimate": 1.089592332950574,
          "median": 0.8937987602171944,
          "ci95": [
            0.9199386889802249,
            1.2946519132094745
          ],
          "n_targets": 55,
          "direction": "lower_is_better"
        },
        "mae_delta": {
          "estimate": 0.17067483879762774,
          "median": 0.14881854651676976,
          "ci95": [
            0.14989006914790645,
            0.19484543316198782
          ],
          "n_targets": 55,
          "direction": "lower_is_better"
        },
        "delta_pearson": {
          "estimate": null,
          "median": null,
          "ci95": [
            null,
            null
          ],
          "n_targets": 0,
          "direction": "higher_is_better"
        },
        "delta_spearman": {
          "estimate": null,
          "median": null,
          "ci95": [
            null,
            null
          ],
          "n_targets": 0,
          "direction": "higher_is_better"
        },
        "raw_cosine": {
          "estimate": null,
          "median": null,
          "ci95": [
            null,
            null
          ],
          "n_targets": 0,
          "direction": "higher_is_better"
        },
        "residual_cosine_vs_train_mean": {
          "estimate": 0.08560267642137051,
          "median": 0.06055699682709543,
          "ci95": [
            0.05268573769351893,
            0.11952193730231264
          ],
          "n_targets": 55,
          "direction": "higher_is_better"
        },
        "pds_l1_raw": {
          "estimate": 0.5133884297520661,
          "median": 0.509090909090909,
          "ci95": [
            0.4360330578512397,
            0.5897520661157024
          ],
          "n_targets": 55,
          "direction": "higher_is_better"
        },
        "effect_size_spearman": {
          "estimate": null,
          "median": null,
          "ci95": [
            null,
            null
          ],
          "n_targets": 55,
          "direction": "higher_is_better",
          "definition": "Spearman correlation across perturbations between predicted and observed delta L2 norms."
        }
      },
      "interpretation": "Comparator evaluated on the identical 55-target aggregate-delta matrix."
    }
  ],
  "pds_scale_sensitivity": {
    "model_id": "safe_ridge_transfer",
    "scale_grid": [
      {
        "prediction_scale": 0.0,
        "pds_l1_raw": 0.5133884297520661
      },
      {
        "prediction_scale": 0.25,
        "pds_l1_raw": 0.5153719008264463
      },
      {
        "prediction_scale": 0.5,
        "pds_l1_raw": 0.5163636363636362
      },
      {
        "prediction_scale": 1.0,
        "pds_l1_raw": 0.5147107438016529
      },
      {
        "prediction_scale": 2.0,
        "pds_l1_raw": 0.5143801652892561
      },
      {
        "prediction_scale": 4.0,
        "pds_l1_raw": 0.5133884297520661
      },
      {
        "prediction_scale": 8.0,
        "pds_l1_raw": 0.5193388429752066
      }
    ],
    "norm_matched_truth_informed": {
      "pds_l1": 0.5431404958677686,
      "ci95": [
        0.46380165289256203,
        0.6195041322314049
      ],
      "n_targets": 55,
      "eligibility": "DIAGNOSTIC_ONLY_TRUTH_DERIVED_SCALING",
      "median_scaling_factor": 14.516575177786605
    },
    "interpretation": "Raw and truth-norm-matched scores are displayed together to reveal scale sensitivity; the norm-matched value is not a fair hidden-test score."
  },
  "locked_metrics": [
    {
      "metric_id": "des_deg_recovery",
      "state": "LOCKED_REQUIRES_CELL_LEVEL_OR_DE_TRUTH"
    },
    {
      "metric_id": "deg_auprc",
      "state": "LOCKED_REQUIRES_CELL_LEVEL_OR_DE_TRUTH"
    },
    {
      "metric_id": "single_cell_distribution_distance",
      "state": "LOCKED_AGGREGATE_PREDICTIONS_ONLY"
    },
    {
      "metric_id": "cell_state_proportion_recovery",
      "state": "LOCKED_AGGREGATE_PREDICTIONS_ONLY"
    }
  ],
  "source_files": [
    {
      "role": "aggregate_prediction_truth_bundle",
      "filename": "prediction_bundle.npz",
      "sha256": "85731e76978206c587074030909b92b060135d5e6e14bcff44dfeeb4d2ec0838"
    },
    {
      "role": "source_result_summary",
      "filename": "result_summary.json",
      "sha256": "610eb57f21a6a432857d669efcc618cbd91bc9d9d147479bd634c3eefeb0f20c"
    },
    {
      "role": "external_target_manifest",
      "filename": "external_target_manifest.tsv",
      "sha256": "77f3918a4f5923972ba57ea23840bad7d86919349c178ad64bab217c896e0abb"
    }
  ],
  "conclusion": "The multi-metric adapter is operational on the existing 55-target external aggregate panel. It does not overturn the frozen NO_DIRECTIONAL_TRANSFER_SUPPORT conclusion and is not evidence of VCC 2026 task compatibility.",
  "claim_boundary": "This is a formula-aligned aggregate-delta diagnostic, not an official cell-eval run, not a VCC 2025 rescore, and not a VCC 2026 task execution. No muscle, DMD, treatment or clinical validity follows from these scores."
}
