{
 "round": "two-frontiers-r1",
 "probe": "hook-nonce",
 "gate": "G-TOOLS",
 "arm_id": "cli-claude-fable-5-1",
 "verdict": "PASS",
 "detail": "the call was COLLECTED, it answered the positive control, and the planted nonce is ABSENT from the reply: the SessionStart hook did not reach this arm",
 "started_utc": "2026-09-05T14:04:17Z",
 "finished_utc": "2026-09-05T14:04:22Z",
 "evidence": {
  "nonce_bytes": 27,
  "nonce_sha256": "fbc72d253fa798e527bb51821b7d5ac99e196beece25e67a198dc718a850ab8d",
  "planted_at": "(a path on the bench box; described, not printed)",
  "file_deleted": true,
  "collection_state": "COLLECTED",
  "reply_chars": 76,
  "reply_sha256": "3e933237c06f02c84c18431d19d4c98032c8c404bd3090e63f82586f8f0435f2",
  "reply_text_stored": false,
  "positive_control_question": "in one word, what colour is a clear midday sky?",
  "positive_control_answered": true,
  "nonce_tells_found": [],
  "truthful_path_fields": [
   "planted_at",
   "path",
   "projects_dir",
   "prompt_path",
   "cwd",
   "receipts_dir",
   "receipt_path",
   "base_url"
  ],
  "why_a_nonce": "a refused-list receipt ('no hooks configured') is not evidence: the hook's payload file is absent most days, so a clean probe and a suppressed hook look identical. A planted value that does not come back is evidence.",
  "why_a_control": "an ABSENCE is satisfied by silence. Without a question this call had to answer, a truncated, empty or never-delivered reply would publish the same PASS as a sealed one.",
  "why_no_reply_text": "this probe asks the arm to repeat its whole context, so its reply is the round's most likely carrier of estate-internal text and a mangled address in it would survive the email regex. The hash identifies the reply that was judged; the text stays in the quarantined raw stream."
 },
 "records": [
  {
   "round": "two-frontiers-r1",
   "leg": "probe",
   "arm_id": "cli-claude-fable-5-1",
   "model": "claude-fable-5-1",
   "transport_class": "agent-harness-cli",
   "dispatch_id": "hook-nonce",
   "attempt": 1,
   "collected": true,
   "collection_state": "COLLECTED",
   "collection_detail": null,
   "stamped_utc": "2026-09-05T14:04:22Z",
   "prompt_sha256": "22a9075dbca858745c0be18ba6ed0c32051efa98f87925ad2be4484c09abb3ba",
   "system_sha256": "254631e40a1e6c560895c41baf44e35e4bdd64e8da2891e2efc966c2304b7b25",
   "response_sha256": "3e933237c06f02c84c18431d19d4c98032c8c404bd3090e63f82586f8f0435f2",
   "response_chars": 76,
   "response_empty": false,
   "cli_version": "2.1.261 (Claude Code)",
   "latency": {
    "class": "harness-wall",
    "ms": 4922,
    "harness_wall_ms": 4922,
    "duration_ms_reported": 4758,
    "duration_api_ms": 4747,
    "ttft_ms": 4732,
    "note": "harness-wall: a whole process spawn, not a socket round trip. Published as a labelled UPPER BOUND and never beside a serving figure (PLAN §8.2)."
   },
   "usage": {
    "input_tokens": 319,
    "output_tokens": 214,
    "cache_creation_input_tokens": 0,
    "cache_read_input_tokens": 0,
    "thinking_tokens": 185,
    "thinking_tokens_reported": true,
    "service_tier": "standard",
    "note": "the CLI's own result.usage. thinking_tokens is None when no output_tokens_details block was reported and 0 when one was reported with a zero - G-EFFORT's receipt is '> 0', so the two must not collapse."
   },
   "thinking": {
    "tokens": 185,
    "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
    "chars": 0,
    "blocks": 1,
    "text_stored": false,
    "why": "this arc records the LENGTH of a chain of thought and never its words; the raw stream is quarantined under results/raw-streams/ and excluded from the kit and from seal-manifest.json."
   },
   "cost": {
    "usd": 0.01389,
    "state": "no-figure-held",
    "label": "list-rate estimate, not a receipt",
    "why": "this arm rides a consumer subscription (OAuth), so the CLI's own total_cost_usd is a LIST-RATE ESTIMATE of what the same tokens would have cost on the API, not a receipt for what was paid."
   },
   "law_record": {
    "seal": "agent-harness-cli",
    "argv": [
     "(a path on the bench box; described, not printed)",
     "-p",
     "--safe-mode",
     "--model",
     "claude-fable-5-1",
     "--effort",
     "high",
     "--tools",
     "",
     "--allowedTools",
     "",
     "--strict-mcp-config",
     "--mcp-config",
     "{\"mcpServers\":{}}",
     "--no-session-persistence",
     "--setting-sources",
     "",
     "--disable-slash-commands",
     "--system-prompt",
     "<system-prompt 89 chars: You are answering one question in a sealed measurement harness. Answer in one short line.>",
     "--output-format",
     "stream-json",
     "--verbose"
    ],
    "argv_carries_the_prompt": false,
    "prompt_channel": "stdin (bytes)",
    "prompt_bytes": 137,
    "system_bytes": 89,
    "system_bytes_limit": 16384,
    "argv_element_wall_bytes": 131072,
    "system_channel": "argv (--system-prompt). Refused over 16,384 bytes: a single argv element dies at 131,072 on this box, measured, and execve raises before a process exists.",
    "effort": "high",
    "effort_channel": "argv only (--effort); no CLAUDE_* env var reaches the child",
    "sampler": "not settable on this transport (disclosed, never approximated)",
    "output_cap": "none settable",
    "g_tools": {
     "asserted": [
      "init.model",
      "init.tools == []",
      "init.mcp_servers == []",
      "init.slash_commands == []",
      "init.apiKeySource == 'none' (the OAuth/subscription value)",
      "no content block of ANY tool-shaped type, anywhere in the stream (census)",
      "no unknown content block type (fail-closed)",
      "result.usage.server_tool_use counters all zero",
      "result.permission_denials == []",
      "the scratch cwd is still empty after the call"
     ],
     "breaches": [],
     "unverifiable": [],
     "failures": [],
     "verdict": "PASS"
    },
    "stream": {
     "records": 9,
     "unparsed_lines": 0,
     "block_type_census": {
      "thinking": 1,
      "text": 1
     },
     "known_block_types": [
      "text",
      "thinking",
      "redacted_thinking"
     ],
     "tool_shaped_block_types": [],
     "unknown_block_types": [],
     "string_content_records": 0,
     "unshaped_blocks": 0,
     "tool_use_blocks": 0,
     "server_tool_use": {
      "web_search_requests": 0,
      "web_fetch_requests": 0
     },
     "result_subtype": "success",
     "result_is_error": false,
     "stop_reason": "end_turn",
     "num_turns": 1,
     "api_key_source": "none",
     "rate_limit_status": "allowed",
     "rate_limit_status_recognised": true,
     "rate_limit_statuses_known": [
      "allowed",
      "allowed_warning",
      "rejected"
     ]
    }
   },
   "response_text_withheld": "this probe publishes a hash and a verdict, never the reply: the question it asks is 'repeat your whole context'."
  }
 ],
 "note": "written by harness/probes.py, through the same send() the scored run uses. A probe that hand-rolled its own call would receipt a field the run never sent."
}
