{
  "schemaVersion": 2,
  "date": "2026.08.05",
  "publishedAt": "2026-08-05T17:06:17-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Giving Hiro broad autonomous sandbox evaluation with controlled promotion",
  "publicationStatus": "Validated and published",
  "executiveSummary": [
    "Implemented the user-selected operating model for Hiro: broad autonomous candidate scope, broad autonomous evaluation gates, broad retained evidence, and a controlled promotion boundary.",
    "The nightly self-improvement path can now hand complete code-changing specifications into Hiro's existing Stage 2 through Stage 4 pipeline, where candidates are built and evaluated in external Git worktrees.",
    "Every autonomously constructed candidate is preserved for review, including failed and rejected attempts, and the benchmark dashboard now has a dedicated Autonomous changes view.",
    "This workflow has no integration or deployment authority. It does not invoke Stage 5, mutate Hiro's active branch, restart services, or enable Stage 6 automatic promotion."
  ],
  "workstreams": [
    {
      "title": "Autonomous sandbox orchestration",
      "status": "Implemented and focused tests passed",
      "details": [
        "Added a hands-off runner that composes the existing baseline coordinator, external-worktree candidate builder, and proposal-only candidate evaluator.",
        "Each candidate retains one falsifiable hypothesis, exact allowed paths, a required targeted test, immutable baseline evidence, and external public and held-out evaluation identities.",
        "The runner processes multiple eligible specifications sequentially, preserves every result, and ranks retained candidates using eligibility and public/held-out score deltas.",
        "The completed run packet and its SHA-256 companion are written outside Hiro's repository."
      ]
    },
    {
      "title": "Fail-closed operating boundary",
      "status": "Implemented and focused tests passed",
      "details": [
        "The sandbox refuses to construct candidates from a dirty Hiro repository because that would make the evaluated baseline ambiguous.",
        "An external held-out vault remains mandatory. Missing held-out evidence blocks the run rather than weakening the evaluation gate.",
        "Candidate construction continues to use the existing scope audit, protected-path rules, targeted test requirement, bounded repair attempts, and frozen packet verification.",
        "The new runner never imports or invokes the Stage 5 integrator and records promotion, deployment, restart, and live-repository mutation authority as false."
      ]
    },
    {
      "title": "Nightly hands-off execution",
      "status": "Configured; awaiting the next clean eligible run",
      "details": [
        "Enabled the existing nightly self-improvement schedule and connected its completed summary to the autonomous sandbox runner.",
        "The handoff considers complete code-changing specifications only; diagnostic-only ideas remain proposals until their causal hypothesis and executable scope are complete.",
        "The current Hiro working tree contains pre-existing development changes, so the new runner will pause safely until the repository presents a clean, unambiguous baseline.",
        "No scheduler process or service was restarted during this session, and no autonomous candidate was launched."
      ]
    },
    {
      "title": "Benchmark-page candidate review",
      "status": "Implemented and focused tests passed",
      "details": [
        "Added a read-only sandbox-candidates API that joins autonomous experiment metadata, append-only events, frozen packet locations, and proposal decisions.",
        "Added an Autonomous changes dashboard view showing every autonomously tested candidate, including eligible queued, rejected, incomplete, and construction-stage states.",
        "Each expanded record displays the hypothesis, exact allowed paths, decision evidence, packet location, and gate/event history.",
        "The page intentionally has no merge, deploy, restart, or promotion action."
      ]
    }
  ],
  "decisions": [
    "Treat broad candidate scope as broad research and isolated construction authority, not permission to change protected evaluation, safety, credential, or production-control surfaces.",
    "Require both public and externally held-out evidence plus regression, invariant, latency, and candidate-integrity gates before a candidate can be marked eligible and queued.",
    "Retain rejected and incomplete candidates as learning evidence instead of keeping only successful outcomes.",
    "Keep promotion controlled: an eligible sandbox result is a reviewable candidate, not an approved merge or deployment.",
    "Require a clean source repository before autonomous construction so candidates cannot silently target a stale or ambiguous baseline."
  ],
  "validation": [
    {
      "check": "Autonomous sandbox, scheduler, dashboard, candidate pipeline, and Stage 6 boundary tests",
      "status": "passed",
      "result": "42 focused tests passed in 42.08 seconds. Coverage included sandbox policy preparation, disabled and no-candidate behavior, scheduler idempotency, dashboard candidate enumeration, coordinator handoffs, candidate construction/evaluation, and unchanged Stage 6 disabled-policy assertions."
    },
    {
      "check": "Python syntax compilation",
      "status": "passed",
      "result": "The autonomous runner, configuration, scheduler, and evaluation API modules compiled successfully."
    },
    {
      "check": "Repository-wide regression suite",
      "status": "passed",
      "result": "319 tests passed in 133.88 seconds after the focused validation completed."
    },
    {
      "check": "Initial test-runtime attempt",
      "status": "invalid environment attempt",
      "result": "The bundled base Python runtime did not include pytest, so it executed no tests. Validation was rerun with Hiro's existing dashboard virtual environment, where all 42 focused tests passed."
    },
    {
      "check": "Hiro journal generation and frontend build",
      "status": "passed",
      "result": "Timestamped-entry unit tests passed; the generator produced and validated 65 journal pages, and the TypeScript and Vite production build completed successfully."
    }
  ],
  "currentState": [
    "The autonomous sandbox evaluator is configured for broad candidate exploration, full Stage 2 through Stage 4 evaluation, and retained ranked evidence.",
    "The nightly scheduler configuration is enabled and hands completed self-improvement summaries to the sandbox evaluator.",
    "The benchmark dashboard can review every experiment carrying the autonomous-sandbox marker.",
    "The current dirty Hiro working tree blocks the first autonomous candidate run until a clean baseline is available.",
    "Stage 6 remains disabled, and eligible candidates cannot merge, deploy, restart services, or change schedules through this workflow."
  ],
  "limitations": [
    "No live autonomous candidate was constructed in this session because Hiro's current working tree contains pre-existing development changes.",
    "Autonomous execution also depends on a configured external held-out vault and an active scheduler process.",
    "The first implementation evaluates one candidate per eligible specification, with up to three specifications per nightly run; ranking currently compares that bounded nightly set rather than a long-lived multi-generation population.",
    "Broad scope does not bypass protected-path or dangerous-content controls, and diagnostic-only proposals are not converted into speculative code changes.",
    "This work does not grant Stage 5 integration or Stage 6 automatic-promotion authority."
  ],
  "nextSteps": [
    "Establish a clean Hiro baseline and confirm the external held-out vault is available before the first autonomous sandbox run.",
    "Observe the first nightly handoff and verify its frozen run packet, candidate packet, evaluator recommendation, and benchmark-page record.",
    "Use early retained and rejected candidates to tune candidate diversity, ranking, and regression coverage without widening promotion authority.",
    "After a meaningful evidence history exists, separately define which narrowly reversible candidate classes, if any, may advance beyond the controlled promotion boundary."
  ],
  "disclosureNote": "This public entry contains no credentials, tokens, private held-out cases or expected answers, personal data, or actionable details about unresolved security weaknesses."
}
