{
  "schemaVersion": 2,
  "date": "2026.08.04",
  "publishedAt": "2026-08-04T09:29:57-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Connecting frozen Hiro candidates to proposal-only evaluation gates",
  "publicationStatus": "Validated and published",
  "executiveSummary": [
    "Hiro's frozen Stage 3 candidate packets can now enter a bounded Stage 4 evaluator that verifies candidate integrity, runs matching public and external held-out suites, executes targeted and regression tests, and applies statistical, category, invariant, and p95-latency gates.",
    "Candidate observations execute from the external candidate worktree in fresh child processes, while evaluation runtime output and recommendation packets remain outside the source repository.",
    "The evaluator writes append-only run, proposal, event, and decision evidence, freezes a SHA-256-backed recommendation, and has no merge, promotion, deployment, restart, scheduler, or cleanup authority.",
    "The repository-wide test suite passed with 277 tests. A dedicated end-to-end validation produced an eligible-for-human-review recommendation with perfect public and held-out pass rates, passing targeted and regression tests, and explicit confirmation that no integration action occurred."
  ],
  "workstreams": [
    {
      "title": "Frozen-candidate and baseline integrity",
      "status": "Implemented and tested",
      "details": [
        "Added verification of the Stage 3 packet hash, frozen status, base commit, worktree HEAD, allowed-path scope, changed-file list, exact file hashes, and immutable snapshots before candidate execution and after all tests.",
        "Required a baseline-ready Step 2 handoff whose pin, variant manifest, source commit, exact public suite hash, exact external held-out suite hash, observation coverage, and visibility labels agree with the append-only evaluation ledger.",
        "Made candidate manifest creation retry-stable by binding its timestamp and code identity to the frozen packet, preventing an interrupted evaluation from conflicting with its own immutable ledger record."
      ]
    },
    {
      "title": "Isolated public and held-out execution",
      "status": "Implemented and tested",
      "details": [
        "Added a worktree process adapter that starts a fresh Python child for each observation and imports the requested entry point from the external candidate worktree.",
        "Kept process runtime data in a separate external directory, disabled bytecode and pytest cache writes, normalized the Windows child-process path environment, and re-audited the candidate after execution.",
        "Used deterministic run identifiers so completed runs are reused and interrupted append-only runs resume without repeating recorded case repetitions."
      ]
    },
    {
      "title": "Proposal-only quality gates",
      "status": "Implemented and tested",
      "details": [
        "Connected both public and held-out reports to score-delta, confidence-separation, category-regression, invariant, and p95-latency checks against the matching pinned baseline.",
        "Added a configurable maximum p95-latency ratio with a default ceiling of 1.10; missing latency evidence now rejects a proposal when the latency gate is enabled.",
        "Ran the candidate's Stage 3 targeted tests together with explicit regression paths and made any regression failure reject an otherwise eligible metric result."
      ]
    },
    {
      "title": "Frozen recommendation and authority boundary",
      "status": "Implemented and tested",
      "details": [
        "Recorded the promotion proposal, experiment event, and public and held-out decisions in the append-only laboratory ledger.",
        "Frozen each recommendation as JSON with a SHA-256 companion outside the repository, including baseline and candidate identities, run metrics, test output, integrity audits, gate reasons, and policy settings.",
        "Limited outcomes to rejected or eligible for human review. Every packet records that merge, promotion, and deployment were not performed and that explicit human approval remains required."
      ]
    }
  ],
  "decisions": [
    "Treat the frozen Stage 3 packet as the candidate identity and reject any packet, worktree, snapshot, changed-file, or scope mismatch before trusting evaluation evidence.",
    "Run candidate behavior from its external worktree rather than importing candidate code into Hiro's long-lived coordinator process.",
    "Require both public and external held-out improvement; success on only one evaluation surface cannot produce an eligible recommendation.",
    "Make latency a first-class proposal gate alongside score, confidence, category, and invariant evidence.",
    "Keep regression testing independent of aggregate evaluation scores so a candidate cannot compensate for a deterministic regression with gains elsewhere.",
    "Preserve Stage 4 as proposal-only. This implementation does not authorize or provide an integration path."
  ],
  "validation": [
    {
      "check": "Focused candidate-evaluator and proposal-gate tests",
      "status": "passed",
      "result": "13 tests passed. Coverage included valid end-to-end evaluation, candidate tampering, suite tampering, regression rejection, latency rejection, invariant and category rejection, no-merge evidence, and interrupted public-run recovery without duplicated observations."
    },
    {
      "check": "Repository-wide Hiro regression suite",
      "status": "passed",
      "result": "277 tests passed in 47.94 seconds."
    },
    {
      "check": "Dedicated external-worktree Step 4 validation",
      "status": "passed",
      "result": "A separately preserved validation ran a frozen bounded candidate from an external worktree. Public and held-out pass rates and weighted scores were 1.0, both targeted and unrelated regression tests passed, the recommendation was eligible for human review, and merge, promotion, and deployment remained false. The frozen recommendation packet SHA-256 was 902f63f34295d271c7e60b902b09300933883bfdd9ae29809fa4e1846b38d125."
    },
    {
      "check": "Source repository integration boundary",
      "status": "passed",
      "result": "The validation fixture source repository retained only its sealed baseline commit and had a clean working tree after candidate evaluation."
    },
    {
      "check": "Hiro journal generation and frontend build",
      "status": "passed",
      "result": "The timestamped-entry tests passed, the generator produced 57 journal pages and passed its schema, identity, alias, sitemap, Atom, noindex, and IndexNow validation, and the TypeScript and Vite production build completed successfully."
    }
  ],
  "currentState": [
    "The Stage 3 builder and Stage 4 evaluator now form a continuous frozen-packet handoff from bounded candidate construction through proposal-only recommendation.",
    "Every candidate must pass integrity checks, matching public and held-out comparisons, deterministic regression tests, invariant checks, and latency policy before it can be labeled eligible for human review.",
    "Interrupted candidate runs can resume from append-only observations with a stable immutable candidate manifest.",
    "No candidate has been merged, promoted, deployed, or used to restart Hiro as part of this work."
  ],
  "limitations": [
    "This session validates the Stage 4 implementation and one bounded end-to-end fixture; it is not a multi-cycle Stage 4 graduation campaign.",
    "The dedicated validation used a deterministic fixture candidate rather than a live Qwen-generated candidate, so real-model proposal quality and repeated operational reliability remain to be measured.",
    "The default 1.10 latency ratio is intentionally conservative and may need evidence-based calibration for model-backed evaluations whose runtime variance is higher than the fixture's.",
    "Eligible recommendations still require explicit human review, and there is deliberately no automated merge or deployment path."
  ],
  "nextSteps": [
    "Run a diverse sequence of real frozen Stage 3 candidates through the Stage 4 evaluator and measure valid completion, eligibility, rejection correctness, interruption recovery, and latency stability.",
    "Include candidates that should fail individual public, held-out, regression, latency, and invariant gates to measure false acceptance and false rejection rather than testing only successful candidates.",
    "Calibrate the latency threshold from repeated comparable runs without weakening the requirement for explicit latency evidence.",
    "Define and satisfy a separate Stage 4 graduation benchmark before considering any reviewed integration workflow."
  ],
  "disclosureNote": "This public entry contains no credentials, tokens, private evaluation cases, personal data, or actionable details about unresolved security weaknesses."
}
