{
  "schemaVersion": 1,
  "contentVersion": "1",
  "reviewedAt": "2026-10-09",
  "origin": "Original fictional AgentDrills scenarios",
  "limitations": "Deterministic authored replays, not live models or a validated assessment.",
  "missions": [
    {
      "id": "false-success",
      "title": "The job that wasn’t done",
      "tag": "Evidence",
      "minutes": 2,
      "summary": "A confident completion message hides a queued job.",
      "goal": "Export the inventory to a CSV and provide a working download link.",
      "contract": "export_csv returns a job ID first. Only status=complete includes a usable download URL.",
      "trace": [
        [
          "01 · CALL",
          "export_csv({ warehouse: \"north\" })"
        ],
        [
          "02 · RESULT",
          "{ job_id: \"job_42\", status: \"queued\" }"
        ],
        [
          "03 · AGENT",
          "“Done — your inventory CSV is ready.”"
        ]
      ],
      "question": "What failed here?",
      "diagnoses": [
        "The export used the wrong warehouse.",
        "The agent treated acceptance as completion.",
        "The job ID is too short."
      ],
      "diagnosis": 1,
      "diagnosisWhy": "The result says queued. It contains neither a completed state nor a download URL.",
      "repairs": [
        {
          "label": "Poll the job to a terminal state, then report its actual result.",
          "ok": true,
          "replay": [
            "get_job(\"job_42\") → { status: \"complete\", download: \"/files/north.csv\" }",
            "Agent → “The export is complete. Here is the CSV.”"
          ],
          "why": "A terminal result supports the completion claim. In a real system, bound the polling and report a pending or failed state honestly."
        },
        {
          "label": "Change the message to “100% complete”.",
          "ok": false,
          "replay": [
            "No new tool result. Job remains queued.",
            "Agent → “100% complete.”"
          ],
          "why": "Stronger wording cannot supply missing evidence."
        },
        {
          "label": "Submit the export again until a job ID arrives.",
          "ok": false,
          "replay": [
            "export_csv(...) → { job_id: \"job_43\", status: \"queued\" }",
            "Two jobs now exist; neither has a verified download."
          ],
          "why": "A second request creates more work without verifying the first."
        }
      ],
      "hint": "Compare the tool’s status with the final claim.",
      "principle": "A tool accepting work is not the same as the work finishing."
    },
    {
      "id": "duplicate-retry",
      "title": "The echoing reservation",
      "tag": "Retries",
      "minutes": 2,
      "summary": "A timeout turns one intended booking into two.",
      "goal": "Reserve exactly one fictional meeting pod for 14:00.",
      "contract": "A timed-out write may have succeeded. reserve_pod supports an idempotency_key and lookup_by_key. Reusing a key returns the original reservation.",
      "trace": [
        [
          "01 · CALL",
          "reserve_pod({ time: \"14:00\", idempotency_key: \"meet-7\" })"
        ],
        [
          "02 · RESULT",
          "Network timeout. Server outcome unknown."
        ],
        [
          "03 · CALL",
          "reserve_pod({ time: \"14:00\", idempotency_key: \"meet-8\" })"
        ],
        [
          "04 · RESULT",
          "{ reservation: \"pod_B\" } · Audit shows pod_A also reserved."
        ]
      ],
      "question": "What made the duplicate possible?",
      "diagnoses": [
        "The retry used a new operation key after an uncertain result.",
        "All timeouts mean the first write failed.",
        "The reservation was made too early."
      ],
      "diagnosis": 0,
      "diagnosisWhy": "A new key identifies a new booking. The timeout did not establish whether the first booking existed.",
      "repairs": [
        {
          "label": "Retry with a third new key.",
          "ok": false,
          "replay": [
            "reserve_pod(..., \"meet-9\") → pod_C",
            "Three reservations now exist."
          ],
          "why": "A new key creates another independent write."
        },
        {
          "label": "Declare the first booking failed and keep the second.",
          "ok": false,
          "replay": [
            "Agent stops. Audit still shows pod_A and pod_B."
          ],
          "why": "Declaring an outcome does not reconcile server state."
        },
        {
          "label": "From the timeout, look up meet-7 and reuse that key if a retry is needed.",
          "ok": true,
          "replay": [
            "lookup_by_key(\"meet-7\") → { reservation: \"pod_A\", status: \"confirmed\" }",
            "Agent reports one confirmed reservation; no new write."
          ],
          "why": "Reconcile an uncertain write using its stable identity. This replay branches before the duplicate; existing duplicates would also need authorized cleanup."
        }
      ],
      "hint": "A timeout is uncertainty, not a rollback.",
      "principle": "Preserve operation identity across retries; reconcile uncertain writes."
    },
    {
      "id": "stale-state",
      "title": "Yesterday’s draft",
      "tag": "State",
      "minutes": 2,
      "summary": "An old read quietly overwrites a teammate’s edit.",
      "goal": "Change only a shared document title to “Launch notes”.",
      "contract": "get_doc returns a version. update_doc can require if_version and change only the title. A version conflict rejects the write without changes.",
      "trace": [
        [
          "01 · READ",
          "get_doc() → { version: 12, title: \"Draft\", body: \"Old notes\" }"
        ],
        [
          "02 · EXTERNAL",
          "Teammate saves version 13 with body: \"Updated notes\"."
        ],
        [
          "03 · CALL",
          "update_doc({ title: \"Launch notes\", body: \"Old notes\" })"
        ],
        [
          "04 · RESULT",
          "{ version: 14, body: \"Old notes\" }"
        ]
      ],
      "question": "Why did the teammate’s work disappear?",
      "diagnoses": [
        "Titles cannot contain spaces.",
        "The new title was too long.",
        "The agent wrote an old full snapshot without a version guard."
      ],
      "diagnosis": 2,
      "diagnosisWhy": "Version 12’s body was written over version 13. The request needed only a title change.",
      "repairs": [
        {
          "label": "Write the same full document again.",
          "ok": false,
          "replay": [
            "update_doc(old snapshot) → version 15",
            "Body remains “Old notes”."
          ],
          "why": "Repeating stale data does not restore the missing update."
        },
        {
          "label": "From the old read, patch the title with if_version=12; on conflict reread and retry narrowly.",
          "ok": true,
          "replay": [
            "update_doc({ title: \"Launch notes\", if_version: 12 }) → conflict; no write",
            "get_doc() → version 13 · body “Updated notes”",
            "update_doc({ title: \"Launch notes\", if_version: 13 }) → version 14; body preserved"
          ],
          "why": "A conditional narrow edit detects changed state and preserves unrelated content. Real conflicts may require a person to resolve intent."
        },
        {
          "label": "Remove version tracking so conflicts cannot happen.",
          "ok": false,
          "replay": [
            "Unconditional write succeeds.",
            "The teammate’s body is still overwritten."
          ],
          "why": "Hiding conflicts removes the signal that state changed."
        }
      ],
      "hint": "Which fields were requested, and which version did the agent read?",
      "principle": "Read fresh state and make the smallest guarded write."
    },
    {
      "id": "missing-constraint",
      "title": "The fast but wrong route",
      "tag": "Constraints",
      "minutes": 2,
      "summary": "The fastest route breaks an explicit requirement.",
      "goal": "Find a cycling route with no unpaved segments.",
      "contract": "Route results report surface. The no-unpaved constraint is mandatory; duration is a preference.",
      "trace": [
        [
          "01 · USER",
          "“A cycling route to the lake, with no unpaved segments.”"
        ],
        [
          "02 · CALL",
          "find_routes({ destination: \"lake\", mode: \"cycle\" })"
        ],
        [
          "03 · RESULT",
          "A: 18 min, mixed paved/gravel · B: 24 min, fully paved"
        ],
        [
          "04 · AGENT",
          "“Choose A — it is the fastest.”"
        ]
      ],
      "question": "Which decision rule was lost?",
      "diagnoses": [
        "Hard constraints must filter candidates before optimizing speed.",
        "Every cycling route must be the longest one.",
        "The result needed an exact distance."
      ],
      "diagnosis": 0,
      "diagnosisWhy": "Route A violates the stated surface requirement even though it is faster.",
      "repairs": [
        {
          "label": "Recommend A but omit its surface.",
          "ok": false,
          "replay": [
            "Agent → “A takes 18 minutes.”",
            "The recommended route still contains gravel."
          ],
          "why": "Omitting a disqualifying fact does not satisfy the user’s constraint."
        },
        {
          "label": "Ask the user again whether gravel is acceptable.",
          "ok": false,
          "replay": [
            "Agent asks a question already answered by “no unpaved segments”.",
            "No compliant route is recommended."
          ],
          "why": "The requirement is clear and a compliant option exists; asking again delays a supported answer."
        },
        {
          "label": "Filter for fully paved routes, then choose B.",
          "ok": true,
          "replay": [
            "Filter surface=fully_paved → [B]",
            "Agent → “B takes 24 minutes and is fully paved.”"
          ],
          "why": "The recommendation meets the hard constraint. If no route qualified, report that rather than quietly weakening the requirement."
        }
      ],
      "hint": "Separate “must have” from “nice to have”.",
      "principle": "Satisfy hard constraints before optimizing preferences."
    },
    {
      "id": "schema-mismatch",
      "title": "A date in the wrong shape",
      "tag": "Tool contracts",
      "minutes": 2,
      "summary": "The tool rejects a plausible-looking argument.",
      "goal": "Read the fictional sensor archive for October 9, 2026.",
      "contract": "read_archive accepts { date: string } where date is YYYY-MM-DD. It is a read-only tool.",
      "trace": [
        [
          "01 · CALL",
          "read_archive({ date: \"10/09/26\" })"
        ],
        [
          "02 · RESULT",
          "Validation error: date must match YYYY-MM-DD. No read performed."
        ],
        [
          "03 · CALL",
          "read_archive({ date: \"10/09/26\" })"
        ],
        [
          "04 · RESULT",
          "Same validation error."
        ]
      ],
      "question": "What should change before another call?",
      "diagnoses": [
        "The sensor should be deleted.",
        "The argument should follow the documented date format.",
        "The tool should be called faster."
      ],
      "diagnosis": 1,
      "diagnosisWhy": "The error is deterministic input validation, so repeating the same argument cannot fix it.",
      "repairs": [
        {
          "label": "Use { date: \"2026-10-09\" } and inspect the result.",
          "ok": true,
          "replay": [
            "read_archive({ date: \"2026-10-09\" }) → { rows: 24, status: \"ok\" }",
            "Agent summarizes the returned archive."
          ],
          "why": "Correct the contract violation using the unambiguous date in the task. Ambiguous dates in real requests require clarification."
        },
        {
          "label": "Retry the same argument five times.",
          "ok": false,
          "replay": [
            "Five calls → five identical validation errors.",
            "No archive has been read."
          ],
          "why": "Retries help some transient failures, not unchanged invalid input."
        },
        {
          "label": "Tell the user there were no sensor readings.",
          "ok": false,
          "replay": [
            "Agent → “No readings.”",
            "Tool never returned archive data."
          ],
          "why": "A rejected request is not evidence of an empty dataset."
        }
      ],
      "hint": "The error describes the accepted input shape.",
      "principle": "Repair deterministic contract errors before retrying."
    },
    {
      "id": "verification-loop",
      "title": "The check after the check",
      "tag": "Stopping rules",
      "minutes": 2,
      "summary": "A completed read-only task gets trapped in repeat checks.",
      "goal": "Report how many items are in an immutable snapshot, using at most two reads.",
      "contract": "Snapshot snap_9 is immutable. count_items returns an authoritative exact count; no second system is involved.",
      "trace": [
        [
          "01 · CALL",
          "count_items({ snapshot: \"snap_9\" })"
        ],
        [
          "02 · RESULT",
          "{ snapshot: \"snap_9\", count: 8, exact: true }"
        ],
        [
          "03 · CALL",
          "count_items({ snapshot: \"snap_9\" }) → 8"
        ],
        [
          "04 · PLAN",
          "“Verify again, then verify that verification.”"
        ]
      ],
      "question": "Why is another check unnecessary in this case?",
      "diagnoses": [
        "Verification is always wasteful.",
        "Eight is always the right answer.",
        "The authoritative immutable result already meets the task’s stopping rule."
      ],
      "diagnosis": 2,
      "diagnosisWhy": "The snapshot cannot change, the tool is authoritative, and a third call would exceed the stated read budget.",
      "repairs": [
        {
          "label": "Start a fresh five-step audit.",
          "ok": false,
          "replay": [
            "Three more identical reads → 8 each.",
            "Read budget exceeded; no new evidence."
          ],
          "why": "More procedure can consume resources without reducing a relevant uncertainty."
        },
        {
          "label": "Report eight items with the snapshot ID and stop.",
          "ok": true,
          "replay": [
            "Agent → “Snapshot snap_9 contains 8 items.”",
            "Total reads: 2. No further tool call."
          ],
          "why": "Stop when the required evidence is sufficient. This does not generalize to mutable state, ambiguous outcomes, or high-impact writes."
        },
        {
          "label": "Skip the result and guess seven.",
          "ok": false,
          "replay": [
            "Agent → “Probably 7.”",
            "Claim contradicts the exact tool result."
          ],
          "why": "Bounded verification still requires using the evidence you already have."
        }
      ],
      "hint": "What uncertainty could one more identical read resolve?",
      "principle": "Define sufficient evidence and stop when you have it."
    },
    {
      "id": "transfer",
      "title": "One badge, one record",
      "tag": "Transfer challenge",
      "minutes": 3,
      "summary": "Apply the principles in a new setting, without being told which one to use.",
      "goal": "Create exactly one event badge labeled “RIVER”. Do not create a duplicate.",
      "contract": "create_badge accepts an operation_key. A timeout may occur after creation. get_by_key reads the final badge. Reusing the same key cannot create another badge. A preview is not a final badge.",
      "trace": [
        [
          "01 · CALL",
          "create_badge({ label: \"RIVER\", operation_key: \"entry-31\" })"
        ],
        [
          "02 · RESULT",
          "Timeout. Creation outcome unknown."
        ],
        [
          "03 · LOCAL",
          "Preview cache: “RIVER” · not a creation receipt."
        ]
      ],
      "question": "Which diagnosis fits the available evidence?",
      "diagnoses": [
        "The preview proves creation succeeded.",
        "Creation is uncertain; retrying with a new key risks a duplicate.",
        "The label must be longer."
      ],
      "diagnosis": 1,
      "diagnosisWhy": "Neither a timeout nor a local preview tells you whether the write completed.",
      "repairs": [
        {
          "label": "Create a new badge with operation_key=\"entry-32\".",
          "ok": false,
          "replay": [
            "create_badge(..., \"entry-32\") → badge_12",
            "Audit: badge_11 already existed for entry-31. Two badges."
          ],
          "why": "A new operation key bypasses the duplicate protection."
        },
        {
          "label": "Report success from the preview cache.",
          "ok": false,
          "replay": [
            "Agent → “Badge created.”",
            "No authoritative creation result has been inspected."
          ],
          "why": "A rendered preview does not establish final state."
        },
        {
          "label": "Look up entry-31; verify its final label, and reuse that key only if creation still needs retrying.",
          "ok": true,
          "replay": [
            "get_by_key(\"entry-31\") → { id: \"badge_11\", label: \"RIVER\", status: \"created\" }",
            "Agent reports badge_11; no extra write."
          ],
          "why": "Combine uncertain-write reconciliation, stable operation identity, and a claim grounded in final state."
        }
      ],
      "hint": "Separate a preview from a receipt, then preserve the identity of the original operation.",
      "principle": "Resolve uncertainty without creating a second action, then report only the verified result."
    }
  ]
}
