{
  "schemaVersion": 2,
  "date": "2026.08.17",
  "publishedAt": "2026-08-17T18:32:32-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Removing text-anchor rejection from candidate construction",
  "publicationStatus": "Implemented, validated, restarted, and exercised live",
  "executiveSummary": [
    "A full funnel audit confirmed a construction-system design flaw: 33 of 33 candidate packets created after the previous restart failed before Stage 4, while no new Stage 4 packet was produced.",
    "Many attempts were rejected because model-supplied old-content text did not match exactly once. Those failures evaluated patch serialization rather than the merit of the proposed behavior.",
    "Revision 48c0427 removes exact-text matching as an admission gate. Candidate edits are now deterministically materialized inside the isolated worktree, then judged by syntax, targeted tests, production guards, baseline contrast, public and held-out evaluation, adversarial tests, and canaries.",
    "The first fresh live candidate completed five materialized patch attempts with zero missing, duplicate, ambiguous, or absent-anchor rejections. It did not promote because its generated implementation broke three production behavior tests."
  ],
  "workstreams": [
    {
      "title": "End-to-end funnel diagnosis",
      "status": "Completed",
      "details": [
        "Inspected the ranked queue, recent lifecycle events, candidate packets, Stage 4 packet creation, and the active source revision.",
        "Found 33 candidate packets created after the earlier restart; all 33 had candidate-failed status and each used five construction attempts.",
        "Found no Stage 4 packet created during that same period, proving ideas were not reaching comparative evaluation.",
        "Observed repeated exact-text failures such as missing or ambiguous modify locators alongside candidates that passed their narrow targeted test but regressed production response-boundary tests."
      ]
    },
    {
      "title": "Non-rejecting candidate text editor",
      "status": "Completed",
      "details": [
        "Added one shared resilient text-edit implementation for the Stage 3 candidate builder and the legacy code-editor path.",
        "When a locator is exact and unique, the editor replaces it normally. When it repeats, the first occurrence is used deterministically.",
        "When a locator is stale or absent, the closest text region is replaced rather than rejecting the candidate. A missing locator appends the proposed content, and a missing allowed target is materialized as a new file.",
        "A stale symbol name falls back to the same resilient text edit. An existing symbol still uses the Python syntax-tree boundary.",
        "Create operations now materialize or overwrite their isolated target instead of being rejected because a target already exists.",
        "Every receipt records the edit-resolution strategy so later diagnosis can distinguish exact, repeated, structural-fallback, append, and create-or-overwrite behavior."
      ]
    },
    {
      "title": "Preserved evaluation authority",
      "status": "Completed",
      "details": [
        "Workspace containment, explicit path allowlists, and evaluator-integrity boundaries remain enforced before a candidate edit can affect files.",
        "Syntax parsing, targeted test collection and execution, production guard tests, baseline contrast, comparative evaluation, adversarial security comparison, and canary checks remain unchanged.",
        "The change removes text serialization refusal; it does not make a malformed or behaviorally regressive patch eligible for promotion."
      ]
    },
    {
      "title": "Live post-restart exercise",
      "status": "Completed",
      "details": [
        "Restarted Hiro on revision 48c0427 with Qwen 3.8 connected and ports 8000, 8001, and 8765 owned by one service process.",
        "The revision change requeued blocked work and the ranked system selected a reproduced arithmetic incident for construction.",
        "The fresh candidate used syntax-tree symbol replacement for the production file and create-or-overwrite for its targeted test on every attempt.",
        "The frozen packet contained zero text-anchor rejection markers across five attempts.",
        "The candidate remained unpromoted because the generated replacement broke three independent production response-boundary tests."
      ]
    }
  ],
  "decisions": [
    "Treat model-supplied text locators as advisory proposal metadata rather than a prerequisite for evaluation.",
    "Always materialize an in-scope text proposal deterministically inside the disposable candidate worktree, even when its locator is stale, repeated, or absent.",
    "Continue rejecting path escapes and evaluator-integrity violations because those define experiment scope rather than text-edit quality.",
    "Let executable validation decide whether the materialized patch is correct; do not infer correctness from successful text replacement.",
    "Retain production guard failures as real functional evidence and do not promote a candidate that passes only its candidate-authored targeted test."
  ],
  "validation": [
    {
      "check": "Focused editor and sandbox suite",
      "status": "passed",
      "result": "36 tests passed in 16.20 seconds."
    },
    {
      "check": "Expanded candidate-pipeline suite",
      "status": "passed",
      "result": "118 tests passed in 137.18 seconds across candidate construction, editing, sandboxing, continuous queue behavior, evaluation, integration, response boundaries, and the continuous governor."
    },
    {
      "check": "Complete Hiro suite",
      "status": "passed",
      "result": "679 tests passed in 200.40 seconds."
    },
    {
      "check": "Duplicate text locator",
      "status": "passed",
      "result": "Regression coverage confirms a repeated locator materializes by replacing the first exact occurrence instead of being rejected."
    },
    {
      "check": "Stale or missing locator",
      "status": "passed",
      "result": "Regression coverage confirms stale locators use the closest text region and missing allowed files are materialized."
    },
    {
      "check": "Stale symbol fallback",
      "status": "passed",
      "result": "Regression coverage confirms a missing symbol name falls back to resilient text materialization instead of terminating construction."
    },
    {
      "check": "Fresh live candidate",
      "status": "functional failure after successful edit materialization",
      "result": "Five attempts completed with no patch-application errors and no text-anchor rejection markers. The final attempt failed three production response-boundary tests."
    },
    {
      "check": "Runtime health",
      "status": "passed",
      "result": "Hiro restarted successfully, reported health ok, and connected to qwen/qwen3.8-27b."
    }
  ],
  "currentState": [
    "Hiro is running on revision 48c0427.",
    "The continuous ranked queue is active and blocked artifacts tied to the older builder revision are eligible for rebuilding.",
    "Text-anchor mismatch no longer prevents an in-scope candidate from reaching executable construction validation.",
    "No candidate was promoted during this session.",
    "The first post-fix candidate failed for a reproducible functional reason: its generated final-gate replacement broke three existing production tests."
  ],
  "limitations": [
    "Removing text-anchor rejection fixes artifact admission but does not improve the local model's ability to design a correct patch.",
    "The first live candidate still replaced a complex response-boundary function too broadly. Candidate implementation quality is now the dominant observed bottleneck.",
    "Best-effort fuzzy placement can materialize an edit in an unintended region. The isolated worktree and executable gates are therefore essential and remain mandatory.",
    "The fresh candidate did not reach Stage 4 because it failed production construction guards, so a complete new promotion has not yet occurred."
  ],
  "nextSteps": [
    "Continue the live queue and measure how many candidates now reach production guards and Stage 4 without text-edit rejection.",
    "Use recorded edit-resolution strategies as a separate construction-funnel metric.",
    "Address broad whole-function replacement as a candidate-generation quality problem if it remains the leading functional failure.",
    "Do not weaken independent production guards merely to increase promotion count; improve patch quality until candidates pass them."
  ]
}
