{
  "artifact": "Shapd AI public Proof Lab evidence sample",
  "generated_for": "customer-facing evaluation demonstration",
  "data_policy": "Synthetic or sanitized data only. Sealed verifier logic and raw trajectories are excluded.",
  "scenarios": [
    {
      "id": "financial-customer-creation",
      "source_pack": "financial-operations-v1",
      "environment": "Stripe-like resettable MCP mock",
      "displayed_runs": [
        { "type": "retained Codex baseline", "reward": 1, "duration_seconds": 17.587 },
        { "type": "intentional duplicate-action control", "reward": 0, "duration_seconds": 1.46 }
      ],
      "pack_validation": {
        "environment_tests": "422/422",
        "scenarios_across_financial_packs": 18,
        "oracle_v1": "12/12 across three consecutive runs",
        "no_action_v1": "0/12",
        "wrong_target_v1": "0/12",
        "duplicate_action_v1": "0/12"
      },
      "interpretation": "Regression and safety evidence. The current baseline passes; this is not presented as frontier difficulty."
    },
    {
      "id": "storage-recovery",
      "source_task": "crash-consistent pack compaction",
      "displayed_runs": [
        { "type": "reviewed recurring model failure", "reward": 0, "failure": "malformed newer candidate did not fall back to older valid location" },
        { "type": "fresh reference validation", "reward": 1 }
      ],
      "validation": {
        "static_checks": "19/19",
        "task_metadata_checks": "28/28",
        "dynamic_checks": "2/2",
        "fresh_oracle": 1,
        "fresh_untouched": 0,
        "reviewed_calibration": "2/8 passes; six genuine code failures"
      },
      "interpretation": "Difficulty evidence for one named task and calibration, not a generalized claim about every model or coding task."
    }
  ]
}
