{
  "title": "Deterministic AI Checklist",
  "version": "1",
  "date": "2026-10-11",
  "publisher": "Deterministic AI, maintained by Kognitos",
  "url": "https://www.deterministicai.org/topics/deterministic-ai-checklist/",
  "scope": "Editorial evaluation worksheet; not a certification, external standard, vendor ranking or executable test harness.",
  "evidence_status_values": [
    "provided",
    "partial",
    "missing"
  ],
  "test_outcome_values": [
    "met",
    "not met",
    "untested"
  ],
  "sources": [
    "https://docs.pytorch.org/docs/main/notes/randomness.html",
    "https://metr.org/blog/2025-06-05-recent-reward-hacking/",
    "https://docs.langchain.com/oss/python/langgraph/use-time-travel",
    "https://docs.temporal.io/workflow-activity"
  ],
  "criteria": [
    {
      "id": "DAI-01",
      "criterion": "Execution boundary",
      "question": "Which result is claimed to be deterministic?",
      "evidence_requested": "Written claim, compared outputs and exclusions",
      "evidence_status": "",
      "test_outcome": "",
      "evidence_reference": "",
      "owner": "",
      "follow_up": ""
    },
    {
      "id": "DAI-02",
      "criterion": "Inputs and state",
      "question": "Can the starting conditions be reconstructed?",
      "evidence_requested": "Input record, retrieved versions, conversation, tool responses and initial state",
      "evidence_status": "",
      "test_outcome": "",
      "evidence_reference": "",
      "owner": "",
      "follow_up": ""
    },
    {
      "id": "DAI-03",
      "criterion": "Version manifest",
      "question": "Which versions and environment are covered?",
      "evidence_requested": "Model, prompt, rule, workflow and runtime versions",
      "evidence_status": "",
      "test_outcome": "",
      "evidence_reference": "",
      "owner": "",
      "follow_up": ""
    },
    {
      "id": "DAI-04",
      "criterion": "Fresh executions",
      "question": "Were comparisons based on fresh computation?",
      "evidence_requested": "Run identifiers, cache settings, reset procedure and raw results",
      "evidence_status": "",
      "test_outcome": "",
      "evidence_reference": "",
      "owner": "",
      "follow_up": ""
    },
    {
      "id": "DAI-05",
      "criterion": "Equality criterion",
      "question": "What counts as an equal result?",
      "evidence_requested": "Comparison rule, tolerances, exclusions, case counts, run counts and differences",
      "evidence_status": "",
      "test_outcome": "",
      "evidence_reference": "",
      "owner": "",
      "follow_up": ""
    },
    {
      "id": "DAI-06",
      "criterion": "Correctness",
      "question": "Was correctness checked independently?",
      "evidence_requested": "Approved reference result or business rule and comparison findings",
      "evidence_status": "",
      "test_outcome": "",
      "evidence_reference": "",
      "owner": "",
      "follow_up": ""
    },
    {
      "id": "DAI-07",
      "criterion": "Exceptions",
      "question": "What happens with missing or conflicting evidence?",
      "evidence_requested": "Exception tests, authorized resolutions and approval/reuse scope",
      "evidence_status": "",
      "test_outcome": "",
      "evidence_reference": "",
      "owner": "",
      "follow_up": ""
    },
    {
      "id": "DAI-08",
      "criterion": "Action confirmation",
      "question": "What proves the intended action occurred?",
      "evidence_requested": "Authorized arguments, destination response, resulting state and repeated-request tests",
      "evidence_status": "",
      "test_outcome": "",
      "evidence_reference": "",
      "owner": "",
      "follow_up": ""
    },
    {
      "id": "DAI-09",
      "criterion": "Replay semantics",
      "question": "Which steps reuse results or execute again?",
      "evidence_requested": "Replay step map and separate replay, retry and fresh-run results",
      "evidence_status": "",
      "test_outcome": "",
      "evidence_reference": "",
      "owner": "",
      "follow_up": ""
    },
    {
      "id": "DAI-10",
      "criterion": "Evaluation integrity",
      "question": "Can a reviewer trust the evaluation record?",
      "evidence_requested": "Raw artifacts, evaluator controls and reference-answer access controls",
      "evidence_status": "",
      "test_outcome": "",
      "evidence_reference": "",
      "owner": "",
      "follow_up": ""
    }
  ]
}
