{
  "schema": "nmd-vcell-model-evaluation-contract/1.0",
  "source_schema": "nmd-vcell-model-evaluation-contract-source/1.0",
  "contract_id": "MODEL-EVALUATION-CONTRACT:RESEARCH-OS:V201",
  "updated": "2026-09-15",
  "local_candidate": {
    "state_id": "LOCAL:RESEARCH_OS:V201",
    "deployment_state": "LOCAL_CANDIDATE_NOT_PRODUCTION"
  },
  "authority_rule": "This contract resolves model admission, metric applicability and generalization coverage from existing released registries. It does not run a model, rescore a benchmark or create a DMD prediction.",
  "adapter_registry": {
    "adapter_count": 5,
    "execution_authorized_count": 0,
    "records": [
      {
        "adapter_id": "ADAPTER:TXPERT:V1",
        "model": "TxPert",
        "interface_state": "HISTORICAL_RUN_ONLY",
        "gateway_decision": "REJECT_UNTIL_NEW_PREREGISTRATION_OR_UNTOUCHED_TRUTH",
        "compatible_task_ids": [
          "G0"
        ],
        "execution_state": "HISTORICAL_LOCAL_GATE_FAIL_NO_REPEAT",
        "released_decision": "STOP_UNTIL_NEW_PREREGISTRATION_OR_UNTOUCHED_TRUTH",
        "claim_ceiling": "E0_EXECUTION_ONLY",
        "receipt_count": 0,
        "source_state": "HISTORICAL_LOCAL_GATE_FAIL_NO_RETRY",
        "execution_authorized": false
      },
      {
        "adapter_id": "ADAPTER:STATE:V1",
        "model": "STATE",
        "interface_state": "HISTORICAL_RECEIPTS_ONLY",
        "gateway_decision": "REJECT_NO_ADVANCEMENT",
        "compatible_task_ids": [
          "G0"
        ],
        "execution_state": "HISTORICAL_RECEIPTS_MIGRATED",
        "released_decision": "NO_ADVANCEMENT",
        "claim_ceiling": "E0_EXECUTION_ONLY",
        "receipt_count": 8,
        "source_state": "HISTORICAL_RUNS_MIGRATED",
        "execution_authorized": false
      },
      {
        "adapter_id": "ADAPTER:STACK:V1",
        "model": "Stack",
        "interface_state": "FORWARD_PATH_SMOKE_ONLY",
        "gateway_decision": "REJECT_NO_TASK_COMPATIBLE_CHECKPOINT",
        "compatible_task_ids": [],
        "execution_state": "LOCAL_NVME_CORE_BASE_FORWARD_PASS_NO_CHECKPOINT",
        "released_decision": "NO_NMD_B1_BENCHMARK_NO_COMPATIBLE_PRETRAINED_GENETIC_CHECKPOINT",
        "claim_ceiling": "E0_EXECUTION_ONLY",
        "receipt_count": 2,
        "source_state": "SOURCE_PINNED_LOCAL_NVME_CORE_PASS",
        "execution_authorized": false
      },
      {
        "adapter_id": "ADAPTER:TAHOE-X1:V1",
        "model": "Tahoe-x1",
        "interface_state": "INPUT_CONTRACT_FAILED",
        "gateway_decision": "REJECT_RAW_COUNTS_REQUIRED",
        "compatible_task_ids": [],
        "execution_state": "SOURCE_PINNED_INPUT_CONTRACT_FAIL",
        "released_decision": "DO_NOT_RUN_ON_LOG_NORMALIZED_INPUT; ACQUIRE_RAW_COUNTS_FIRST",
        "claim_ceiling": "E0_SOURCE_ONLY",
        "receipt_count": 2,
        "source_state": "SOURCE_PINNED_INPUT_CONTRACT_AUDIT",
        "execution_authorized": false
      },
      {
        "adapter_id": "ADAPTER:SLIM:V1",
        "model": "SLIM",
        "interface_state": "PRIMARY_SOURCE_PENDING",
        "gateway_decision": "REJECT_SOURCE_NOT_VERIFIED",
        "compatible_task_ids": [],
        "execution_state": "PRIMARY_SOURCE_PENDING",
        "released_decision": "NOT_ELIGIBLE_FOR_EXECUTION",
        "claim_ceiling": "E0_SOURCE_ONLY",
        "receipt_count": 0,
        "source_state": "PRIMARY_SOURCE_PENDING",
        "execution_authorized": false
      }
    ]
  },
  "gateway": {
    "gateway_id": "MODEL-GATEWAY:NMDVCELL:V1",
    "mode": "CONTRACT_ONLY_NO_HOSTED_INFERENCE",
    "default_decision": "REJECT",
    "execution_endpoint_available": false,
    "dmd_prediction_route_open": false,
    "required_gates": [
      "registered task contract",
      "source revision and licence identity",
      "exact input contract pass",
      "task-compatible weights or checkpoint",
      "immutable split and feature mapping",
      "permanent baseline comparison",
      "metric applicability and independent-unit contract",
      "receipt, uncertainty and claim-ceiling record"
    ],
    "unregistered_request_policy": "REJECT_NO_REGISTERED_TASK_AND_MODEL_RUN",
    "current_dmd_decision": "ABSTAIN_NO_CALIBRATED_DMD_MODEL_RUN",
    "released_model_run_count": 6,
    "calibrated_dmd_model_run_count": 0,
    "matched_dmd_candidate_response_truth": "MISSING"
  },
  "metric_lab": {
    "comparison_rule": "A metric is comparable only within the same task, split, feature space, endpoint and independent-unit contract. Missing metrics stay locked and no global score is computed.",
    "global_score_permitted": false,
    "families": [
      {
        "family_id": "AGGREGATE_ERROR",
        "label": "Aggregate error",
        "metric_ids": [
          "RMSE",
          "MAE"
        ],
        "state": "EXECUTED_BOUNDED",
        "eligible_task_ids": [
          "G0",
          "G1"
        ],
        "boundary": "Aggregate response vectors only; lower error does not establish direction, disease transfer or function."
      },
      {
        "family_id": "RESPONSE_DIRECTION",
        "label": "Response direction",
        "metric_ids": [
          "raw cosine",
          "residual cosine",
          "delta Pearson",
          "delta Spearman"
        ],
        "state": "EXECUTED_BOUNDED",
        "eligible_task_ids": [
          "G0",
          "G1"
        ],
        "boundary": "Same-context and external aggregate diagnostics remain separate and do not establish DMD transport."
      },
      {
        "family_id": "PERTURBATION_RETRIEVAL",
        "label": "Perturbation retrieval",
        "metric_ids": [
          "PDS L1 raw"
        ],
        "state": "DIAGNOSTIC_SENSITIVITY_ONLY",
        "eligible_task_ids": [
          "G1"
        ],
        "boundary": "Formula-aligned aggregate diagnostic; not an official cell-eval or Virtual Cell Challenge score."
      },
      {
        "family_id": "SAMPLING_UNCERTAINTY",
        "label": "Sampling uncertainty",
        "metric_ids": [
          "paired bootstrap interval",
          "paired sign-flip test",
          "target bootstrap interval"
        ],
        "state": "EXECUTED_BOUNDED",
        "eligible_task_ids": [
          "G0",
          "G1"
        ],
        "boundary": "Intervals quantify the registered benchmark unit only, not donor, disease-stage or clinical uncertainty."
      },
      {
        "family_id": "CELL_LEVEL_DE_DISTRIBUTION",
        "label": "Cell-level and DE distribution",
        "metric_ids": [
          "DES",
          "DEG AUPRC",
          "single-cell distribution distance",
          "cell-state proportion recovery"
        ],
        "state": "LOCKED_CELL_LEVEL_OR_DE_TRUTH_MISSING",
        "eligible_task_ids": [],
        "boundary": "Aggregate predictions cannot be relabelled as cell-level or differential-expression truth."
      },
      {
        "family_id": "CALIBRATION_ABSTENTION",
        "label": "Calibration and abstention",
        "metric_ids": [
          "calibration",
          "coverage",
          "selective risk",
          "abstention quality"
        ],
        "state": "PARTIAL_DIAGNOSTIC_NOT_DMD_CALIBRATION",
        "eligible_task_ids": [
          "G1"
        ],
        "boundary": "Historical selective-risk diagnostics do not constitute a calibrated DMD predictor."
      },
      {
        "family_id": "FUNCTIONAL_OUTCOME",
        "label": "Functional and safety outcome",
        "metric_ids": [
          "functional hit rate",
          "toxicity miss rate",
          "replication",
          "synergy classification"
        ],
        "state": "LOCKED_MATCHED_OUTCOME_MISSING",
        "eligible_task_ids": [],
        "boundary": "No directly measured candidate-conditioned DMD functional outcome exists."
      }
    ]
  },
  "generalization_cube": {
    "cube_id": "GENERALIZATION-CUBE:NMDVCELL:G0-G7:V1",
    "representation": "SPARSE_REGISTERED_TASK_CELLS_NOT_CARTESIAN_SCORE",
    "axes": [
      "perturbation novelty",
      "biological context",
      "donor or line",
      "dataset or laboratory",
      "disease background",
      "modality or combination"
    ],
    "cells": [
      {
        "task_id": "G0",
        "primary_axis": "perturbation novelty",
        "coordinate": "SAME_CONTEXT_HELD_OUT_TARGET",
        "label": "Same-context target holdout",
        "state": "EXECUTED_LIMITED",
        "ladder_state": "DEBUG_ONLY_NOT_PRIMARY",
        "split_contract": "repeated-fold-v2.2 target holdout",
        "endpoint": "Held-out 2,000-feature mean response vector",
        "independent_unit": "held-out perturbation target; cells do not increase the target denominator",
        "primary_metrics": [
          "RMSE",
          "raw cosine",
          "residual cosine",
          "strict 16-outcome gate"
        ],
        "current_result": "Small average-error gain; strict gate 0/16 and direction unreliable",
        "decision_blocker": "No external cell-level or disease-context truth",
        "route": "/resource/benchmarks/core/"
      },
      {
        "task_id": "G1",
        "primary_axis": "dataset or laboratory",
        "coordinate": "EXTERNAL_UNSEEN_PERTURBATION",
        "label": "External unseen perturbation",
        "state": "EXECUTED_PARTIAL_UNSUPPORTED",
        "ladder_state": "IMPLEMENTED_HEPG2",
        "split_contract": "new perturbation in an external aggregate object",
        "endpoint": "Directional-transfer assessment",
        "independent_unit": "study/dataset-defined biological unit; must be recorded by each ModelRun",
        "primary_metrics": [
          "formula-aligned aggregate directional diagnostics"
        ],
        "current_result": "No directional-transfer support",
        "decision_blocker": "No harmonized cell-population outcome or matched feature space",
        "route": "/resource/benchmarks/external-transfer/"
      },
      {
        "task_id": "G2",
        "primary_axis": "perturbation novelty",
        "coordinate": "UNSEEN_GENE_FAMILY_OR_PATHWAY",
        "label": "Unseen gene family or pathway",
        "state": "REGISTERED_NOT_EXECUTED",
        "ladder_state": "REGISTERED_SPLIT_DATA_NOT_EXECUTED",
        "split_contract": "held-out related targets with leakage-safe exposure audit",
        "endpoint": "Held-out family response distribution",
        "independent_unit": "study/dataset-defined biological unit; must be recorded by each ModelRun",
        "primary_metrics": [
          "DES",
          "PDS",
          "MAE",
          "calibration",
          "abstention"
        ],
        "current_result": "No result",
        "decision_blocker": "Current candidate set is too small for a powered family holdout",
        "route": "/resource/benchmarks/generalization/"
      },
      {
        "task_id": "G3",
        "primary_axis": "donor or line",
        "coordinate": "UNSEEN_DONOR",
        "label": "Unseen donor",
        "state": "LOCKED_NO_TRUTH",
        "ladder_state": "REGISTERED_DONOR_RESOLVED_TRUTH_MISSING",
        "split_contract": "entire donor withheld",
        "endpoint": "Donor-conditioned response and responder distribution",
        "independent_unit": "study/dataset-defined biological unit; must be recorded by each ModelRun",
        "primary_metrics": [
          "calibration",
          "donor replication",
          "distribution distance",
          "abstention"
        ],
        "current_result": "No result",
        "decision_blocker": "No qualifying donor-resolved DMD perturbation matrix",
        "route": "/resource/benchmarks/generalization/"
      },
      {
        "task_id": "G4",
        "primary_axis": "biological context",
        "coordinate": "UNSEEN_CELL_STATE",
        "label": "Unseen cell state",
        "state": "LOCKED_NO_TRUTH",
        "ladder_state": "REGISTERED_MATCHED_TRUTH_MISSING",
        "split_contract": "train in one myogenic state and test another",
        "endpoint": "State-conditioned population transition",
        "independent_unit": "study/dataset-defined biological unit; must be recorded by each ModelRun",
        "primary_metrics": [
          "state occupancy",
          "DES",
          "PDS",
          "MAE",
          "transport distance"
        ],
        "current_result": "No result",
        "decision_blocker": "Current myogenic trajectory is real but unperturbed",
        "route": "/resource/trajectory/"
      },
      {
        "task_id": "G5",
        "primary_axis": "dataset or laboratory",
        "coordinate": "FUTURE_BATCH_OR_EXTERNAL_LAB",
        "label": "Future batch or external laboratory",
        "state": "LOCKED_NO_TRUTH",
        "ladder_state": "REGISTERED_HARMONIZED_TRUTH_MISSING",
        "split_contract": "prospective time and site shift",
        "endpoint": "Prospective response, uncertainty and abstention",
        "independent_unit": "study/dataset-defined biological unit; must be recorded by each ModelRun",
        "primary_metrics": [
          "calibration",
          "AUPRC",
          "hit enrichment",
          "abstention quality"
        ],
        "current_result": "No result",
        "decision_blocker": "No registered prediction or future outcome",
        "route": "/resource/prediction_registry"
      },
      {
        "task_id": "G6",
        "primary_axis": "disease background",
        "coordinate": "INDEPENDENT_DMD_DISEASE_LINE",
        "label": "Independent DMD disease line",
        "state": "LOCKED_NO_TRUTH",
        "ladder_state": "REGISTERED_DMD_PERTURBATION_TRUTH_MISSING",
        "split_contract": "new disease line or genotype",
        "endpoint": "Molecular, functional, toxicity and replication outcomes",
        "independent_unit": "study/dataset-defined biological unit; must be recorded by each ModelRun",
        "primary_metrics": [
          "functional hit rate",
          "AUPRC",
          "calibration",
          "toxicity miss rate",
          "replication"
        ],
        "current_result": "No result",
        "decision_blocker": "0 independent DMD candidate-perturbation outcomes",
        "route": "/resource/minimum-perturbome/"
      },
      {
        "task_id": "G7",
        "primary_axis": "modality or combination",
        "coordinate": "NEW_MODALITY_OR_COMBINATION",
        "label": "New modality and combination",
        "state": "LOCKED_NO_TRUTH",
        "ladder_state": "REGISTERED_OBSERVED_DOUBLE_TRUTH_MISSING",
        "split_contract": "KO, CRISPRa, drug or double perturbation",
        "endpoint": "Main effects, interaction, synergy class and uncertainty",
        "independent_unit": "study/dataset-defined biological unit; must be recorded by each ModelRun",
        "primary_metrics": [
          "interaction error",
          "synergy classification",
          "calibration",
          "toxicity"
        ],
        "current_result": "Only an additive no-interaction preview is available",
        "decision_blocker": "No observed candidate doubles or matched DMD drug screens",
        "route": "/resource/perturbation-lab"
      }
    ],
    "scalar_collapse_permitted": false,
    "counts": {
      "total": 8,
      "executed": 2,
      "registered_not_executed": 1,
      "locked_no_truth": 5
    }
  },
  "source_objects": {
    "adapter_status": {
      "path": "/resource/api/v1.1/model_execution_status_registry.json",
      "sha256": "0e2d0e25be4e421eca5d8b411c8e1292e745e1736d594699ef2870c8b160433e"
    },
    "model_runs": {
      "path": "/resource/api/v1.1/model_run_registry.json",
      "sha256": "a429808f22b477ff67b1f5b84ef5c5dce082401e5c44f79fdb2fabda6539d850"
    },
    "task_contracts": {
      "path": "/resource/api/v1.1/benchmark_task_registry_v2.json",
      "sha256": "1b831ac73203486fc239f52f7a86a5765d8a88380d58a43052bafa95e00e2917"
    },
    "generalization_lineage": {
      "path": "/resource/api/v1.1/generalization_ladder.json",
      "sha256": "7a4749e145dfec23b21b77455bbf9c98f0dc8cb047793fd7e911093f511d9181"
    },
    "metric_diagnostic": {
      "path": "/resource/api/v1.1/vcc_metric_diagnostic.json",
      "sha256": "f4ba29fa92c945ed99bf30f50b7775a613863c416104ac4dd94c2afa91a6a944"
    },
    "route_execution": {
      "path": "/resource/api/v1.1/model_route_execution_v01.json",
      "sha256": "3c4b8f393649e81d674dd50b964e7a09119f16e4b5d24d768f63baccbca76ac7"
    }
  },
  "claim_boundary": "The Model Evaluation Gateway reports admission, metric and truth availability from released objects. It creates no inference endpoint, new benchmark result, DMD prediction, experiment authorization or biological claim.",
  "does_not_imply": [
    "a runnable hosted model",
    "a newly eligible adapter",
    "a global model score",
    "DMD generalization",
    "clinical or therapeutic validity"
  ],
  "new_model_run_count": 0,
  "new_scientific_claim_count": 0,
  "model_evaluation_contract_sha256": "05934cbd7df873b91cfe6290e82e65cf2226a438abf16049a7565a8bcd24745d"
}
