{
  "tool_id": "art-489-model-test-battery",
  "kernel_id": "art-489-model-test-battery",
  "display_name": "Model Test Battery",
  "tool_version": "1.0.0",
  "mandate_type": "compliance_control",
  "purpose": "Runs the deterministic-given-data quantitative model validation battery: discriminatory power (Gini coefficient, Kolmogorov-Smirnov statistic) from scored outcomes, population and characteristic stability (PSI, CSI) between two declared snapshots using the caller's own bins, back-test outcome-vs-predicted per declared bin, and a calibration comparison (predicted vs actual rate, max absolute diff). Every test is graded against a POLICY-SUPPLIED threshold object -- this node never chooses or hardcodes a threshold. If a threshold is missing for a requested test, that test is reported skipped_no_threshold, never silently defaulted; if the underlying data is missing or insufficient (e.g. a single-class scored set for Gini/KS), the test is reported skipped_insufficient_data. Each result carries the test's standard name, its metric value, its threshold, and pass/breach so the artifact reads as a workpaper section, not an opinion. Model inputs and specifications are expected to arrive as a workbook export via the shipped WB-BRIDGE-1 workbook-to-OCG artifact bridge (tool 554); this node itself accepts plain JSON so any upstream source can feed it. Pairs with art-488 model-replication-diff (recompute-and-diff) as the two deterministic legs of a model validation cycle -- neither node emits a soundness or fitness-for-use opinion. Inline deterministic transcendental math (no engine Math.exp/log) so the same input always produces the same execution_hash on every surface. NaN-safe. Zero network, zero PII.",
  "control_description": "Runs the deterministic-given-data quantitative model validation battery: discriminatory power (Gini coefficient, Kolmogorov-Smirnov statistic) from scored outcomes, population and characteristic stability (PSI, CSI) between two declared snapshots using the caller's own bins, back-test outcome-vs-predicted per declared bin, and a calibration comparison (predicted vs actual rate, max absolute diff). Every test is graded against a POLICY-SUPPLIED threshold object -- this node never chooses or hardcodes a threshold. If a threshold is missing for a requested test, that test is reported skipped_no_threshold, never silently defaulted; if the underlying data is missing or insufficient (e.g. a single-class scored set for Gini/KS), the test is reported skipped_insufficient_data. Each result carries the test's standard name, its metric value, its threshold, and pass/breach so the artifact reads as a workpaper section, not an opinion. Model inputs and specifications are expected to arrive as a workbook export via the shipped WB-BRIDGE-1 workbook-to-OCG artifact bridge (tool 554); this node itself accepts plain JSON so any upstream source can feed it. Pairs with art-488 model-replication-diff (recompute-and-diff) as the two deterministic legs of a model validation cycle -- neither node emits a soundness or fitness-for-use opinion. Inline deterministic transcendental math (no engine Math.exp/log) so the same input always produces the same execution_hash on every surface. NaN-safe. Zero network, zero PII.",
  "declared_inputs": [],
  "declared_outputs": [],
  "kernel_digest": "sha256:20e07ad532c59a2c5619ce03881ce808c367002a5572176618bbb8b6ba06e178",
  "trust_label": "deferred -- deterministic source published, zkVM proof not yet generated",
  "data_vintage": "2026-07-10",
  "last_validated": "2026-07-10",
  "conformance_fixtures_vendored": true,
  "compute_proof_ready": "deferred",
  "wave": 77,
  "source_url": "https://ainumbers.co/chaingraph/art-489-model-test-battery.html",
  "generated_at": "2026-07-27T13:08:52.746Z"
}
