{
  "schemaVersion": 2,
  "date": "2026.08.17",
  "publishedAt": "2026-08-17T10:42:58-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Separating functional evaluation from narrow adversarial security gates",
  "publicationStatus": "Implementation completed and validated; Hiro remains paused",
  "executiveSummary": [
    "Hiro's candidate pipeline no longer treats ordinary programming features, response invariants, scope mismatches, or construction errors as terminal safety failures.",
    "Candidate patches remain expressive inside isolated experiment scope. Functional quality is decided by targeted proof plus public and held-out baseline-versus-candidate comparisons.",
    "Security now has a separate baseline-versus-candidate adversarial gate focused on prompt injection, unauthorized execution and tool use, untrusted-content handling, credential and network boundaries, and related boundary bypasses.",
    "The implementation is committed at revision 4d7730b. A focused suite passed 109 tests, a later focused confirmation passed 81 tests, and the complete Hiro suite passed 675 tests in 197.41 seconds. Hiro was not restarted."
  ],
  "workstreams": [
    {
      "title": "Removal of broad static safety rejection",
      "status": "Completed",
      "details": [
        "Removed content-pattern rules that automatically elevated patches merely for using subprocess APIs, file writes, dynamic language features, or network mutation APIs.",
        "Retained immutable automation-integrity boundaries for the evaluation kernel, promotion governor, and the policy that protects those boundaries.",
        "Legacy patch application now relies on explicit workspace scope and integrity validation rather than a coarse high-risk label that prevented otherwise testable code."
      ]
    },
    {
      "title": "Expressive candidate construction",
      "status": "Completed",
      "details": [
        "Removed the incident-specific arithmetic patch recipe that required one exact edit shape at one source line.",
        "Separated construction requirements from security constraints in the candidate request. Construction requirements describe valid patches and attributable tests; security constraints address hostile instructions and genuinely adversarial behavior.",
        "Scope, syntax, test collection, response quality, and implementation failures remain repairable or artifact-blocking functional outcomes rather than being mislabeled as security failures."
      ]
    },
    {
      "title": "Baseline-comparative adversarial gate",
      "status": "Completed",
      "details": [
        "Added a dedicated security-test list to candidate evaluation and execute the identical tests on a detached baseline worktree and the candidate worktree.",
        "A candidate receives a terminal security classification only when the baseline passes and the candidate fails, or when replicated prompt-injection category evidence shows a candidate regression.",
        "If both baseline and candidate fail, the result is inconclusive and blocks promotion without accusing the candidate of causing a security regression. If the candidate repairs a failing baseline, the security comparison records an improvement.",
        "The default adversarial set covers prompt-injection filtering, untrusted external content, credential and network boundaries, unauthorized tool execution, approval scope, and state-mutation boundaries."
      ]
    },
    {
      "title": "Queue and agenda outcome classification",
      "status": "Completed",
      "details": [
        "Changed terminal classification to consume explicit security evidence instead of scanning generic failure text for words such as authority, invariant, unsafe, policy, or scope.",
        "Renamed the benchmark outcome from safety rejected to security rejected so the dashboard reports the narrower meaning accurately.",
        "Applied the same security comparison to open-problem candidate replications and require security eligibility in their paired aggregate decision."
      ]
    }
  ],
  "decisions": [
    "Let candidate models produce normal code within the experiment's declared scope; do not ban common implementation mechanisms by syntax alone.",
    "Judge ordinary behavior, latency, formatting, correctness, and invariants as functional baseline-versus-candidate evidence.",
    "Reserve security rejection for reproducible adversarial regressions involving prompt injection, unauthorized execution or tool use, credential or network boundaries, or an equivalent boundary bypass.",
    "Keep evaluator and promotion-authority files immutable to candidates because allowing candidates to edit their own judge would invalidate the comparison rather than demonstrate an upgrade.",
    "Do not restart Hiro automatically after this policy correction."
  ],
  "validation": [
    {
      "check": "Focused pipeline suite",
      "status": "passed",
      "result": "109 tests passed in 73.63 seconds across candidate construction, evaluation, autonomous sandboxing, continuous queue decisions, integrity policy, benchmark reporting, open-problem execution, and active-loop policy."
    },
    {
      "check": "Complete Hiro suite",
      "status": "passed",
      "result": "675 tests passed in 197.41 seconds."
    },
    {
      "check": "Post-expansion focused confirmation",
      "status": "passed",
      "result": "81 tests passed in 57.67 seconds after the default adversarial set was expanded."
    },
    {
      "check": "Ordinary implementation syntax",
      "status": "passed",
      "result": "A regression test confirms subprocess and file-write code in an ordinary production patch remains a moderate-impact candidate and is not statically rejected as a security violation."
    },
    {
      "check": "Explicit security regression",
      "status": "passed",
      "result": "A fixture with a passing baseline security test and failing candidate security test was rejected with explicit security evidence and frozen baseline and candidate receipts."
    },
    {
      "check": "Functional classification",
      "status": "passed",
      "result": "Regression tests confirm that invariant and allowlist language alone remains repairable functional evidence rather than becoming a terminal security result."
    }
  ],
  "currentState": [
    "Hiro source revision 4d7730b contains the functional and security gate separation.",
    "The Hiro working tree is clean.",
    "The continuous Hiro service remains paused; only the local model service remains available.",
    "No candidate was promoted during this implementation session."
  ],
  "limitations": [
    "Passing adversarial tests cannot prove the absence of every security weakness; the gate establishes non-regression against the configured threat cases.",
    "Candidate generation quality remains dependent on the local proposal model. This change removes false policy barriers but does not guarantee that the model will produce a useful patch.",
    "The security test list is intentionally narrow. Functional defects outside those adversarial categories are still capable of blocking promotion through the normal comparison gates.",
    "The corrected system has been validated by automated tests but has not yet been exercised on a new live candidate after the pause."
  ],
  "nextSteps": [
    "Review the fixed policy and then restart Hiro when authorized.",
    "Observe the first new candidate end to end and verify that construction, functional evaluation, security comparison, canaries, and dashboard classification each produce separate receipts.",
    "Measure construction yield and promotion yield without adding incident-specific patch instructions.",
    "Expand adversarial security coverage only in response to concrete threat-model evidence, while keeping ordinary functional evaluation broad and baseline comparative."
  ]
}
