{
  "schemaVersion": 2,
  "date": "2026.08.17",
  "publishedAt": "2026-08-17T16:06:59-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Restarting Hiro and exercising the functional-security split",
  "publicationStatus": "Hiro restarted; live queue and adversarial comparison validated",
  "executiveSummary": [
    "Hiro was restarted on revision 4d7730b with Qwen 3.8 connected, all three service ports owned by one Hiro process, and the continuous ranked queue active.",
    "Two newly generated live candidates failed during construction and were routed back through functional revision handling. Neither was mislabeled as a security rejection, and neither was promoted.",
    "To exercise the new security gate independently of candidate-generation variance, the current evaluator re-evaluated a previously frozen real candidate using its preserved public and held-out reports. The identical adversarial suite passed 117 tests on the detached baseline and 117 tests on the candidate.",
    "That controlled comparison finished eligible for controlled integration with a passed security status and a functional failure class. It was an evaluator-only test and did not integrate or promote the historical candidate."
  ],
  "workstreams": [
    {
      "title": "Process restart and health verification",
      "status": "Completed",
      "details": [
        "Validated and used the checked-in Windows launcher, including its process-scoped Path normalization and pinned Python runtime.",
        "Confirmed one Hiro process owned ports 8000, 8001, and 8765 after restart.",
        "Confirmed the dashboard and continuous-improvement API responded successfully and the health endpoint reported the model connection as qwen/qwen3.8-27b."
      ]
    },
    {
      "title": "Live continuous-queue exercise",
      "status": "Completed with candidate revisions pending",
      "details": [
        "The queue selected a reproduced arithmetic response failure and created a new baseline packet, isolated worktree, and frozen candidate packet.",
        "The candidate's targeted arithmetic check passed, but three unrelated production response-boundary checks regressed. The queue correctly treated this as a repairable functional construction failure and scheduled revision rather than issuing a security rejection.",
        "A second live candidate for follow-up continuity also exhausted its bounded construction attempt without reaching the evaluation or security gate. It remained a functional construction outcome.",
        "At the final check the ranked system contained 249 ideas, including 173 actionable ideas, one active candidate, and 78 retrying items. The legacy workflow remained read-only."
      ]
    },
    {
      "title": "Controlled adversarial baseline comparison",
      "status": "Passed",
      "details": [
        "Selected a previously frozen candidate that had already completed construction successfully, avoiding a new model-generated patch as a confounding variable.",
        "Reused the candidate's preserved public and held-out reports, then ran the current evaluator's default adversarial test set against both a detached baseline and the candidate worktree.",
        "The suite covered operational boundaries, agent-harness boundaries, untrusted external idea sources, and internet-observation boundaries.",
        "Both sides passed all 117 adversarial tests. The frozen recommendation reported security status passed, regression passed, and eligible for controlled integration.",
        "No integration was requested or performed; this was a proposal-stage evaluator validation only."
      ]
    },
    {
      "title": "Runtime security observation",
      "status": "Observed",
      "details": [
        "The service received unsolicited probes for credential-like resources during the session. The requests were denied with unauthorized responses.",
        "No sensitive request details, credentials, or unresolved exploit information are included in this public entry."
      ]
    }
  ],
  "decisions": [
    "Keep the live ranked queue running after validation so ordinary upgrade work continues in parallel.",
    "Classify patch correctness, response invariants, test failures, and construction failures as functional evidence unless an explicit adversarial baseline-versus-candidate regression is demonstrated.",
    "Use the same adversarial tests on baseline and candidate; only a candidate-caused regression can justify the terminal security-rejected outcome.",
    "Use a preserved real candidate for the first direct security-gate exercise because both newly generated candidates failed earlier in construction and could not test that stage.",
    "Do not promote a historical candidate merely to validate the evaluator. Eligibility and integration remain separate decisions."
  ],
  "validation": [
    {
      "check": "Launcher validation and child-process startup",
      "status": "passed",
      "result": "The checked-in launcher validated successfully, started Hiro, and the resulting service process owned ports 8000, 8001, and 8765."
    },
    {
      "check": "Runtime health",
      "status": "passed",
      "result": "The health response was ok with the LLM connected as qwen/qwen3.8-27b; benchmark and continuous-improvement endpoints returned successfully."
    },
    {
      "check": "First new live candidate",
      "status": "functional revision scheduled",
      "result": "Its targeted arithmetic test passed, but three production response-boundary tests failed. The candidate was not promoted and was not security-rejected."
    },
    {
      "check": "Second new live candidate",
      "status": "construction failed",
      "result": "The bounded candidate builder did not produce a validation-clean patch, so the candidate did not enter global or adversarial evaluation."
    },
    {
      "check": "Detached baseline adversarial suite",
      "status": "passed",
      "result": "117 tests passed in 2.59 seconds with return code 0."
    },
    {
      "check": "Candidate adversarial suite",
      "status": "passed",
      "result": "117 tests passed in 2.16 seconds with return code 0."
    },
    {
      "check": "Frozen evaluator recommendation",
      "status": "passed",
      "result": "The recommendation status was eligible_for_controlled_integration, security status was passed, regression was passed, and failure class was functional."
    },
    {
      "check": "Outcome ledger",
      "status": "verified",
      "result": "The API reported 3 implemented, 1 construction-failed, 31 evaluation-rejected, 4 hypothesis-disproven, and 0 security-rejected outcomes at the final observation."
    }
  ],
  "currentState": [
    "Hiro remains running on source revision 4d7730b.",
    "Qwen 3.8 remains connected and the continuous queue remains active.",
    "One arithmetic candidate was active at the final check, with its next action set to construct an isolated candidate.",
    "No new candidate was promoted during this session.",
    "The Hiro working tree was clean at the final check."
  ],
  "limitations": [
    "The direct security-gate exercise used a previously frozen candidate and preserved evaluation reports. It proves that the current comparative gate runs and classifies that candidate correctly, but it is not a fresh end-to-end promotion.",
    "Both new live candidates failed during construction before reaching global functional comparison or adversarial security evaluation. Candidate-generation quality is therefore the immediate observed bottleneck.",
    "A passing configured adversarial suite demonstrates non-regression for those threat cases, not the absence of every possible security weakness.",
    "The running queue has many historical retry items. Its yield should be measured over multiple fresh candidates before drawing conclusions about promotion rate."
  ],
  "nextSteps": [
    "Allow the active queue to continue producing candidates and inspect the next validation-clean patch end to end.",
    "Measure construction success, evaluation eligibility, security eligibility, canary success, and promotion as separate funnel stages.",
    "Improve candidate construction prompts or repair strategy using recurring failure evidence if construction remains the dominant bottleneck.",
    "Keep adversarial coverage focused on prompt injection, unauthorized execution or tool use, untrusted content, credential and network boundaries, and equivalent boundary bypasses."
  ]
}
