{
  "schemaVersion": 2,
  "date": "2026.08.15",
  "publishedAt": "2026-08-15T22:37:26-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "RSI obstacles become recoverable, testable evidence",
  "publicationStatus": "Implemented, validated, and live",
  "executiveSummary": [
    "Hiro no longer treats every failed candidate transaction as evidence that an idea was bad. The queue now distinguishes construction failure, evaluation rejection, safety rejection, infrastructure blockage, disproven hypotheses, legacy outcomes, supersession, and implementation.",
    "Candidate repair attempts now restart from a clean isolated baseline, receive the prior failure evidence, and must produce a complete alternative candidate. Python candidates can use AST-bounded top-level symbol replacement instead of fragile text matching.",
    "Continuous candidates now prove targeted improvement by running the candidate-authored test against an untouched baseline: it must execute and fail assertions there, then pass on the candidate. Public and held-out suites enforce global non-regression, invariants, categories, and repeated cross-suite latency evidence.",
    "A shadow replay found eleven retained candidates that satisfy the corrected targeted and global evidence contract. Four were rejected because baseline already passed, and twenty-four remained inconclusive rather than receiving false credit.",
    "A real Qwen3.8 retry of a previously failed memory candidate produced a candidate-ready packet after one clean repair. All 597 repository tests passed and the live circuit breaker remains closed."
  ],
  "workstreams": [
    {
      "title": "Persistent candidate recovery",
      "status": "Completed",
      "details": [
        "Each non-terminal builder repair now performs a validated hard reset and untracked-file cleanup inside the exact external candidate worktree before asking for a new strategy.",
        "Repair prompts state that the workspace is clean and require a complete candidate rather than an incremental patch against discarded code.",
        "The builder supplies file-existence state, compact prior validation evidence, exact relevant symbol context, and instructions that distinguish create, modify, append, and symbol replacement operations.",
        "AST-bounded replace_symbol changes only one explicitly named top-level Python function or class inside an allowlisted file and then passes through the existing patch-safety and scope checks."
      ]
    },
    {
      "title": "Targeted improvement with global non-regression",
      "status": "Completed",
      "details": [
        "Continuous RSI candidates use a dedicated evidence mode rather than the general laboratory's global-superiority policy.",
        "The evaluator creates a temporary detached baseline worktree, copies only the candidate's targeted tests into it, and requires an executable assertion failure. A passing baseline proves no improvement; collection or usage failures are inconclusive.",
        "The same tests must pass on the candidate alongside the stable regression suite.",
        "Public and held-out score may vary by at most one weighted case, category regression remains capped at 0.03, candidate invariant failures remain forbidden, and latency rejection requires the regression to repeat in both suites.",
        "The general promotion policy remains global superiority by default; the continuous policy and all thresholds are explicit in the active policy and frozen proposal evidence."
      ]
    },
    {
      "title": "Failure taxonomy and observability",
      "status": "Completed",
      "details": [
        "Frozen builder attempts are summarized into specific obstacle classes including stale anchors, duplicate creation, empty plans, patch-application failures, scope rejection, and safety rejection.",
        "Queue API records now expose outcome_category and aggregate outcome_counts without rewriting historical evidence.",
        "The Ranked Ideas Rejected view displays category totals and each card carries its derived outcome badge.",
        "The live 120 rejected records resolve to 86 construction failures, 28 evaluation rejections, and six legacy rejections; none are silently relabeled as successful."
      ]
    },
    {
      "title": "Safe historical shadow replay",
      "status": "Completed",
      "details": [
        "Added a reusable command that replays retained sandbox evidence against the corrected global policy and optionally runs targeted baseline contrasts.",
        "Shadow replay never promotes, changes queue state, or rewrites old decisions.",
        "Of 74 candidate-ready historical packets, 39 passed corrected global non-regression. Eleven also demonstrated executable targeted baseline failures, four already passed on baseline, and 24 produced inconclusive baseline collection results."
      ]
    }
  ],
  "decisions": [
    "Interpret persistence as trying new bounded strategies and improving the experiment, never bypassing authority, safety, held-out, invariant, or scope constraints.",
    "Keep historical rejection records append-only and derive better classifications on read.",
    "Do not promote historical shadow-replay candidates automatically; use them only to validate selectivity and inform future explicit trials.",
    "Require baseline tests to execute and fail assertions. Missing imports or collection failures are inconclusive, even if the candidate later collects successfully.",
    "Require latency degradation in both public and held-out measurements before treating it as repeatable; a one-suite spike remains evidence but not a rejection by itself.",
    "Retain the stricter global-superiority policy for general laboratory experiments while using targeted proof plus global non-regression for narrow continuous repairs."
  ],
  "validation": [
    {
      "check": "Focused recovery suite",
      "status": "passed",
      "result": "Eighty-seven focused tests passed for clean repairs, AST symbol replacement, builder diagnostics, queue taxonomy, continuous policy, targeted baseline contrast, latency confirmation, shadow replay, dashboard rendering, and active policy loading."
    },
    {
      "check": "Historical shadow replay",
      "status": "passed",
      "result": "Replayed 74 candidate-ready packets without promotion or queue mutation: 39 cleared global non-regression, eleven proved targeted improvement, four already passed on baseline, 24 were inconclusive, and zero contrast executions raised infrastructure errors."
    },
    {
      "check": "Real Qwen3.8 construction retry",
      "status": "passed",
      "result": "A previously failed typed-memory hypothesis was rebuilt from the same 8e19fd0 baseline. The first candidate failed validation; one clean repair returned a complete alternative whose syntax, collection, and targeted tests passed, producing a candidate-ready frozen packet."
    },
    {
      "check": "Authoritative full Hiro suite",
      "status": "passed",
      "result": "All 597 tests passed in 186.75 seconds in Hiro's real pinned environment."
    },
    {
      "check": "Live restart and API",
      "status": "passed",
      "result": "Hiro restarted at commit a0b456f, ports 8001 and 8765 returned on one validated process, eight actionable ideas were admitted, and the circuit breaker remained closed with zero failures."
    },
    {
      "check": "Live dashboard",
      "status": "passed",
      "result": "The Rejected filter visibly displayed construction, evaluation, safety, and disproven-hypothesis totals plus classified historical cards."
    }
  ],
  "currentState": [
    "Hiro is live on Qwen3.8 with one active candidate and seven waiting ideas from the newly admitted batch.",
    "The active candidate will use clean-workspace repairs, specific obstacle evidence, and the corrected continuous evaluation policy.",
    "The historical 120 rejection total remains intact but is no longer presented as one undifferentiated scientific outcome.",
    "The repository is clean at a0b456f and the promotion circuit breaker is closed."
  ],
  "limitations": [
    "The real-model recovery test validates candidate construction, not Stage 4 eligibility or live integration; those remain governed by their later gates.",
    "Twenty-four historical candidates had baseline collection failures and are intentionally inconclusive. A future candidate must write a baseline-executable targeted test to receive improvement credit.",
    "One successful recovery candidate does not establish a stable builder-ready rate; the new queue batch must provide longitudinal evidence.",
    "Global non-regression permits one weighted-case variation to accommodate evaluator granularity and stochasticity, but category and invariant hard gates can still reject that candidate."
  ],
  "nextSteps": [
    "Observe the newly active candidate through construction, targeted contrast, global evaluation, Stage 5 integration, and canary probation.",
    "Measure the new builder-ready rate against the prior 24.6 percent baseline and classify every remaining obstacle.",
    "Convert recurring inconclusive baseline-test patterns into candidate-authoring feedback so future tests execute on both baseline and candidate.",
    "Retry only deduplicated high-value historical construction failures as new append-only lineages after the new live batch establishes stable behavior."
  ]
}
