{
  "schemaVersion": 2,
  "date": "2026.08.06",
  "publishedAt": "2026-08-06T07:29:55-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Completing Hiro's first snapshot-bound internet evaluation replay",
  "publicationStatus": "Validated and published",
  "executiveSummary": [
    "Clarified the Internet observations dashboard so raw snapshots explicitly require no approval, while approved offline evaluation cases are displayed as separate, reviewable decisions.",
    "Converted the reviewed PyPI httpx observation into one immutable evaluation case containing the approved finding, task, assertions, snapshot binding, review record, and companion SHA-256 files.",
    "Ran Hiro's pinned baseline and an identical no-change candidate through the local tool-free evaluator. Both consumed the same frozen snapshot-content hash and independently returned the expected stable version, 0.28.1.",
    "The complete 384-test Hiro suite passed before the service was restarted on implementation commit 5400ac5. Live browser verification then confirmed zero pending approvals, one approved case, a completed replay, identical evidence, and two passing results.",
    "Internet observation and Stage 6 remain disabled. The replay grants no internet, state-changing, promotion, deployment, or Stage 6 authority."
  ],
  "workstreams": [
    {
      "title": "Explicit approval boundary",
      "status": "Completed",
      "details": [
        "Added dashboard language explaining that retained observations are evidence only and never create a decision obligation by themselves.",
        "Added separate counts for pending approvals and approved cases, plus per-snapshot labels distinguishing observation-only evidence from evidence attached to an approved case.",
        "Added a dedicated snapshot-bound evaluation-case panel showing the approved finding, task, assertions, snapshot hash, replay state, answers, and replay integrity hash.",
        "Kept the authority boundary visible in both API and page data: approval covers offline case creation and replay only, never promotion or Stage 6."
      ]
    },
    {
      "title": "Tamper-evident reviewed case",
      "status": "Completed",
      "details": [
        "Extended the review record to preserve the exact approved prompt, assertions, category, repetition count, timeout, reviewer, review time, snapshot ID, packet hash, and sanitized-content hash.",
        "Added an independent SHA-256 companion for the review record in addition to the existing case and snapshot companions.",
        "Added bundle verification that fails closed if the case, approval record, or snapshot binding differs, or if either record attempts to grant promotion or Stage 6 authority.",
        "The approved case is web-replay-c84ec11a10d2ada3 and is bound to sanitized-content SHA-256 4f3f5b971cdf11b1e578e2ea1b630fd34d316d8b9604c59babb91b0b4b8fd979."
      ]
    },
    {
      "title": "Offline baseline and control replay",
      "status": "Completed",
      "details": [
        "Pinned both manifests to Hiro commit 5400ac5ee973daef10d1f850d2749b29d2bf0771 and the same local model, using an explicitly labeled no-change candidate as the control.",
        "The evaluator used no browser, shell, web-search, memory, history, remote fallback, or external request; frozen snapshot content was treated as untrusted evidence rather than instructions.",
        "Both runs returned 0.28.1 and passed the approved exact-output and forbidden-content assertions.",
        "The baseline and candidate reports preserved the identical snapshot binding. The immutable replay artifact and companion verify as SHA-256 46e411814bff085e6130114da5a4726a739d38cf1fb1a4dbb95948b6f064c57a."
      ]
    },
    {
      "title": "Live benchmark deployment and verification",
      "status": "Completed",
      "details": [
        "Restarted only the validated Hiro API processes through the checked-in pinned launcher after committing the implementation.",
        "The service became ready in 1.656 seconds and resumed its three expected listeners.",
        "The live read-only API reports three verified snapshots, zero pending approvals, one approved case, zero case-integrity failures, zero replay-integrity failures, and one completed replay.",
        "Browser inspection verified the rendered approval explanation, all three snapshot states, the expanded case details, identical-hash indicator, two passing results, answers, and replay hash."
      ]
    }
  ],
  "decisions": [
    "Do not make every raw observation an approval item; only a concrete proposed evaluation case can require a future decision.",
    "Treat the user's prior instruction to proceed with the first offline candidate as approval for this bounded PyPI stable-version case only.",
    "Use an identical no-change candidate for the first replay so this session validates evidence binding and evaluation plumbing without implying that a code improvement was discovered.",
    "Freeze completed replay evidence with full manifests, reports, zero-authority declarations, and a SHA-256 companion.",
    "Display autonomously tested changes and snapshot-derived evaluation evidence on the benchmark page while retaining separate semantics for code candidates and source observations.",
    "Leave both internet observation and Stage 6 disabled after the replay."
  ],
  "validation": [
    {
      "check": "Focused observation and dashboard tests",
      "status": "passed",
      "result": "All 61 focused tests passed, including review-companion tampering, reviewed-bundle verification, frozen replay verification, API approval state, and dashboard contract coverage."
    },
    {
      "check": "Complete Hiro suite",
      "status": "passed",
      "result": "All 384 tests passed in 115.00 seconds."
    },
    {
      "check": "Frozen offline replay",
      "status": "passed",
      "result": "Baseline and no-change candidate both passed with answer 0.28.1, identical snapshot-content SHA-256 4f3f5b971cdf11b1e578e2ea1b630fd34d316d8b9604c59babb91b0b4b8fd979, and no evaluation error."
    },
    {
      "check": "Replay integrity and authority",
      "status": "passed",
      "result": "The frozen replay verifies against companion SHA-256 46e411814bff085e6130114da5a4726a739d38cf1fb1a4dbb95948b6f064c57a and records false for internet use, state-changing action, promotion authority, and Stage 6 authority."
    },
    {
      "check": "Live API and browser-rendered page",
      "status": "passed",
      "result": "The live API and page both show three verified snapshots, zero pending approvals, one approved case, a completed replay, identical evidence, and passing baseline and control results."
    },
    {
      "check": "Service and stop state",
      "status": "passed",
      "result": "Hiro is healthy on its expected listeners; startup logs contain no error or rollback evidence; internet observation and Stage 6 DISABLED sentinels remain present."
    },
    {
      "check": "Journal generation and frontend build",
      "status": "passed",
      "result": "The timestamped-entry unit tests passed, 74 journal pages and aliases were validated, and the production frontend build completed successfully."
    }
  ],
  "currentState": [
    "Hiro is running commit 5400ac5ee973daef10d1f850d2749b29d2bf0771 on the existing development branch.",
    "The benchmark page shows three verified observations, zero pending approvals, and one approved snapshot-bound evaluation case.",
    "The first real frozen-snapshot offline replay is complete and tamper-evident; both baseline and the no-change control passed.",
    "The replay is plumbing evidence, not evidence of an improvement, because baseline and candidate intentionally use the same code commit.",
    "Internet observation and Stage 6 remain disabled, and automatic promotion remains disabled."
  ],
  "limitations": [
    "One successful extraction case does not establish that Hiro can reliably turn arbitrary real-world failures into useful evaluations.",
    "The first replay used an unchanged control, so it demonstrates repeatability and isolation but does not measure an actual candidate improvement.",
    "The current dashboard displays approvals and evidence but does not yet provide a general operator workflow for accepting or rejecting future proposed cases.",
    "Observation remains limited to the previously approved Phase 1 sources and requires a separately authorized, low-budget capture session."
  ],
  "nextSteps": [
    "Use the clarified dashboard state as the operator record: no action is needed for observation-only snapshots, and future decisions should be attached to concrete proposed cases.",
    "Add the first genuine snapshot-derived failure case when an observation exposes a reproducible baseline weakness, then test a substantive candidate against the exact same hash.",
    "Accumulate several reviewed cases across small sessions before considering scheduled observation or any expansion of the allowlist.",
    "Keep promotion human-gated and Stage 6 disabled until repeated real candidate comparisons demonstrate reliability beyond the no-change control."
  ],
  "disclosureNote": "This public entry contains no credentials, private held-out cases, personal data, raw retrieved page content, local runtime paths, or actionable details about unresolved security weaknesses."
}
