{
  "schemaVersion": 2,
  "date": "2026.08.11",
  "publishedAt": "2026-08-11T15:25:12-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Hiro adopts one continuous improvement loop with visible scoring and a stable governor",
  "publicationStatus": "Implemented, tested, promoted through live acceptance, hardened, and reactivated",
  "executiveSummary": [
    "Hiro's layered Stage 6B, Stage 6C, passive-corroboration, day-lab, and night-lab improvement machinery has been replaced for automatic operation by one continuously advancing queue.",
    "Every safe external observation now enters a transparent ranked backlog, receives an active bounded investigation, and either closes with evidence or advances through isolated candidate construction, tests, canary checkpoints, and a stable promotion governor.",
    "Routine low-risk improvements use the user's standing authority and report outcomes afterward. High-risk, irreversible, credential, authorization, and governor changes remain outside standing authority.",
    "The benchmark page now presents current work, ranked category scores and totals, completed implementations, explicit rejections, and audited controls that let the user boost or nominate an idea without bypassing gates.",
    "A deliberately unsafe acceptance seed was rejected by its targeted test without changing the active branch. A positive seed completed a real 30-minute canary, survived an intentional Hiro restart with its checkpoint state intact, passed the complete repository gate after an infrastructure repair, and was promoted by fast-forward."
  ],
  "workstreams": [
    {
      "title": "Unified continuous queue",
      "status": "Completed",
      "details": [
        "Added a durable SQLite queue with append-only events, an exclusive runner lease, finite infrastructure retries, and explicit queued, investigating, candidate, canary, implemented, and rejected states.",
        "Removed passive 'awaiting local corroboration' behavior. Each supported idea is actively tested against a bounded local worker surface; an unreproducible idea closes with a reason instead of waiting indefinitely.",
        "The scheduler advances the queue on every poll independently of discovery cadence, permits only one candidate build and one canary at a time, and can investigate the next ranked item while the current canary waits.",
        "External discovery remains prompt-safe input to hypothesis generation only. It cannot authorize commands, widen paths, alter the governor, or promote code.",
        "Historical terminal records were retained, while the remaining actionable ideas were migrated into the new queue."
      ]
    },
    {
      "title": "Stable promotion governor",
      "status": "Completed",
      "details": [
        "Introduced a small policy-bound governor that is independent of generated worker candidates and is itself excluded from autonomous modification.",
        "The governor accepts only low- or moderate-risk requests with a complete timed canary, an unchanged base revision, an exact direct-child candidate commit, a clean active tree, an exact changed-file set, and no protected paths, deletions, or symbolic links.",
        "Before promotion it runs the complete repository test suite in the isolated candidate worktree.",
        "A passing candidate reaches the live branch only by fast-forward. No merge commit or rebase is permitted, and a pre-promotion reference is retained for recovery.",
        "Rollback is additive rather than history-rewriting. Irreversible actions, secrets, credentials, authorization policy, and governor changes remain outside standing authority."
      ]
    },
    {
      "title": "Transparent idea scoring",
      "status": "Completed",
      "details": [
        "Each idea receives visible scores for task success, autonomy, capability, reliability, tool quality, memory, efficiency, improvement throughput, and simplicity.",
        "The weighted impact score is combined with reach and confidence and offset by cost, risk, and complexity. The formula version, category weights, factors, impact total, base priority, and user boost are all returned by the API.",
        "A user boost adds a bounded interest signal; 'make next' nominates one queued idea. Neither action skips investigation, tests, canary time, or governor review.",
        "Priority actions are recorded as append-only audit events."
      ]
    },
    {
      "title": "Benchmark page realignment",
      "status": "Completed",
      "details": [
        "Replaced obsolete nightly, autonomous-change, and comparison navigation with current improvement, ranked ideas, evaluation health, performance, and evidence history.",
        "Added a live pipeline view for queue, investigation, construction, testing, canary, and final promotion or rejection.",
        "Added current-canary details, score cards with category and factor breakdowns, implemented and rejected outcome sections, affected-file reporting, and audited priority controls.",
        "The live HTTP API and observatory page both returned successfully after restart, and the served page contained the new ranked-idea interface."
      ]
    },
    {
      "title": "Curiosity review quality",
      "status": "Completed",
      "details": [
        "Added a versioned worker policy for corroborated curiosity reviews and a quality probe that independently measures source safety and whether a finding explains its relevance to the user.",
        "The initial policy requires one to five corroborating sources and intentionally omitted the relevance explanation to provide a real positive improvement opportunity.",
        "External observations continue through prompt-injection screening and local evidence checks before they can become candidates."
      ]
    },
    {
      "title": "Legacy Stage 6 retirement",
      "status": "Completed",
      "details": [
        "Disabled automatic day-lab, night-lab, Stage 6B, and Stage 6C schedulers in the active policy.",
        "Archived the exact-revision Stage 6C enablement after the redesign changed the repository revision.",
        "Preserved old campaign, decision, and validation evidence as read-only history instead of deleting it.",
        "The continuous governor is now the sole active-branch promotion authority for routine worker improvements."
      ]
    },
    {
      "title": "Acceptance-discovered infrastructure hardening",
      "status": "Completed",
      "details": [
        "The first final-gate attempt exposed that deterministic candidate worktrees lacked Hiro's ignored pinned runtime, causing two launcher tests to fail for environmental reasons after all four canary probes passed.",
        "The missing runtime junction was provisioned without changing the candidate, and the governor reran the complete suite against the same candidate revision and completed canary evidence.",
        "The repair was then made permanent so future deterministic candidates provision the pinned runtime before testing.",
        "The declared systemic circuit breaker is now enforced: three consecutive infrastructure failures stop queue advancement and discovery instead of creating an endless retry loop.",
        "A new regression covers the circuit-open behavior, and the complete hardened repository suite passed."
      ]
    }
  ],
  "decisions": [
    "Treat improvement as a measurable change in user outcomes rather than compliance with a narrow implementation lane.",
    "Replace passive corroboration waits with active, bounded local experiments and explicit terminal reasons.",
    "Keep idea discovery, candidate generation, and promotion authority separate so untrusted content cannot become authority.",
    "Use standing user authority for reversible low-risk worker changes and reserve explicit review for high-risk or authority-changing work.",
    "Use a real 30-minute low-risk canary with checkpoints at minutes zero, five, fifteen, and thirty.",
    "Continue investigating the next ranked idea while a canary waits, but do not construct a second candidate concurrently.",
    "Expose every score, file impact, test gate, and outcome in one user-facing benchmark workflow."
  ],
  "validation": [
    {
      "check": "Focused redesign suite",
      "status": "passed",
      "result": "26 queue, governor, curiosity-policy, scheduler-adapter, and dashboard tests passed."
    },
    {
      "check": "Full repository regression",
      "status": "passed",
      "result": "516 tests passed in 159.18 seconds."
    },
    {
      "check": "Negative acceptance seed",
      "status": "passed",
      "result": "A candidate that removed source corroboration failed its targeted policy test, closed as rejected, and left the active revision unchanged."
    },
    {
      "check": "Restart recovery",
      "status": "passed",
      "result": "Hiro was intentionally stopped and restarted during the positive canary; the same start time and completed minute-zero checkpoint remained in durable state."
    },
    {
      "check": "Unattended queue advance",
      "status": "passed",
      "result": "While the positive canary waited, the next ranked production idea advanced through investigation to a parked candidate state without user authorization."
    },
    {
      "check": "Dashboard priority control",
      "status": "passed",
      "result": "A live boost and clear action both updated the score, recorded audit events, and reported that no gate bypass occurred."
    },
    {
      "check": "Positive 30-minute canary and promotion",
      "status": "passed after infrastructure repair",
      "result": "Minutes zero, five, fifteen, and thirty passed across an intentional restart. The first final-suite attempt found a missing worktree runtime dependency; after provisioning it, the exact same candidate passed all 516 tests and fast-forwarded successfully."
    },
    {
      "check": "Post-acceptance hardening regression",
      "status": "passed",
      "result": "27 focused tests passed, followed by 517 complete repository tests in 158.75 seconds."
    },
    {
      "check": "Final live service and dashboard",
      "status": "passed",
      "result": "Hiro restarted hidden on the hardened revision; the promoted relevance policy was active, the circuit was closed, the positive and negative outcomes appeared correctly, and the next production idea was already in candidate state."
    }
  ],
  "currentState": [
    "Hiro is running in the background and its local health, queue API, and benchmark page respond successfully.",
    "Eleven migrated or seeded ideas are visible in the durable queue, including preserved terminal history.",
    "The negative seed is rejected, the positive seed is implemented, and another production idea is already in candidate state.",
    "Legacy Stage 6 evidence is retained read-only and its live exact-revision enablement has been archived.",
    "Routine user approval is no longer required for low-risk candidates, but every candidate still passes independent tests, a timed canary, and the stable governor.",
    "The systemic circuit breaker is closed with zero consecutive infrastructure failures."
  ],
  "limitations": [
    "General externally inspired candidates still depend on the local model producing a coherent bounded patch that clears the established evaluator.",
    "High-risk or irreversible changes remain intentionally outside autonomous standing authority.",
    "The synchronous final repository gate can temporarily reduce local API responsiveness for roughly the duration of the test suite; this is visible as canary verification rather than an absent workflow.",
    "The journal and benchmark omit credentials, private response content, held-out prompts, and actionable security detail."
  ],
  "nextSteps": [
    "Monitor implemented and rejected outcomes on the redesigned benchmark page and use its priority controls when a particularly interesting idea deserves attention.",
    "Allow the already-started production candidate to reach an evidence-backed implementation or rejection without manual authorization.",
    "Measure whether relevance explanations improve the usefulness of future curiosity reviews, and let a later candidate revise the policy if the live evidence disagrees.",
    "Review early autonomous candidates for outcome quality and tune score weights only from observed results rather than adding new process layers."
  ],
  "disclosureNote": "This public entry omits credentials, private response content, held-out prompts, machine-local paths, external post text, and security-sensitive implementation detail."
}
