{
  "schemaVersion": 2,
  "date": "2026.07.23",
  "publishedAt": "2026-07-23T08:04:38-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Step 3 calibration repair completed",
  "publicationStatus": "Published",
  "executiveSummary": [
    "Implemented and launched the finite Step 3 diversity controller instead of using the prior unbounded repetition loop. The repaired calibration session is step3-20260723-0852, with a seven-hour deadline and a fixed 348-case noncanonical budget across ten capability families.",
    "The opening Qwen canonical gate passed 6/6 in 110.526 seconds (stateful-20260723T145256Z, qwen_qwen3.6-35b-a3b). The fresh repaired P1 completed 40 cases across development, blind, and canary partitions with 37/40 passes (92.5%), zero infrastructure errors, zero invariant failures, 100% canary, and 91.7% blind performance.",
    "The repair exposed complete allowed-choice vocabularies without identifying the correct choice, corrected an accidentally double-escaped scoring regex, and added executable regression coverage proving every generated oracle accepts its canonical answer. Earlier invalid or interrupted attempts remain immutable and excluded."
  ],
  "workstreams": [
    {
      "title": "Finite controller and sealed portfolio",
      "status": "Completed",
      "details": [
        "Added the hiro.benchmarks.development_cycle.step3 controller with prepare, validate-plan, record-preflight, run, status, and close commands.",
        "Sealed six phases totaling 348 noncanonical cases: 40 calibration, 80 breadth, 60 evolving state, 60 cross-family, 48 adaptive reserve, and 60 blind frontier. The portfolio covers calendar temporal reasoning, task replanning, email grounding, cross-tool coordination, approval lifecycle, concurrency, recovery, memory continuity, source judgment, and productive autonomy.",
        "Validation admitted 348 unique structural signatures, a 100% generated novelty rate, every required family, a maximum family share below 15%, and a maximum topology share of 0.57%."
      ]
    },
    {
      "title": "Oracle and retry integrity",
      "status": "Completed",
      "details": [
        "Reworked generated prompts so expected answer tokens are not written into the prompt. Their required response varies from the generated state, such as the only safe calendar slot, blocked dependency, approval state, version freshness, or recovery state.",
        "Sealed a distinct immutable VariantManifest per phase and partition during preparation, preventing retry attempts from changing the ledger manifest solely because the current time changed.",
        "Added regression coverage for finite balanced preparation, idempotent prepare/close, passing canonical preflight recording, and expected-answer non-leakage."
      ]
    },
    {
      "title": "P1 calibration repair",
      "status": "Completed",
      "details": [
        "Recorded the passing opening canonical report as immutable preflight evidence for fresh session step3-20260723-0852.",
        "The repaired P1 completed through the one-time local Windows task: 22 development, 12 blind, and 6 canary observations; 37 passed, three failed, and no evaluation or invariant error occurred.",
        "No production connector, live calendar, email, task system, reading list, schedule, or external service was changed."
      ]
    }
  ],
  "decisions": [
    "Do not accept superficial signature diversity when expected labels are exposed by the prompt; correct the oracle contract before scoring a model.",
    "Treat interrupted evaluation execution as a controller reliability requirement. Retry must reuse the exact sealed variant rather than create a fresh immutable record.",
    "Run only P1 after readiness validation; do not release the full campaign merely because a timebox exists.",
    "Retain previously prepared and interrupted session records as audit evidence, but exclude them from valid evaluation counts."
  ],
  "validation": [
    {
      "check": "Focused Step 3 and legacy development-cycle tests",
      "status": "passed",
      "result": "7 tests passed in 1.73 seconds, including complete choice vocabulary, non-leaking output instructions, and executable canonical-regex acceptance for every generated case."
    },
    {
      "check": "Campaign manifest validation",
      "status": "passed",
      "result": "step3-20260723-0852 validates with 348 cases, 348 unique structural signatures, all ten families, 100% generated novelty, and 0.29% maximum topology share."
    },
    {
      "check": "Opening canonical preflight",
      "status": "passed",
      "result": "stateful-20260723T145256Z used qwen_qwen3.6-35b-a3b, passed 6/6, and completed in 110.526 seconds."
    },
    {
      "check": "Repaired P1 calibration",
      "status": "passed",
      "result": "Fresh session step3-20260723-0852 passed 37/40 (92.5%): development 20/22, blind 11/12, canary 6/6. There were zero observation errors and zero invariant failures. The three misses were unrelated memory-correction, quoted-email, and conservative-autonomy decisions."
    }
  ],
  "currentState": [
    "The active repaired campaign session is step3-20260723-0852 with its finite manifest and immutable P1 evidence recorded.",
    "P1 now satisfies the campaign decision table for advancement: fresh pass rate is above 90%, canary is perfect, blind exceeds 90%, and the three misses are not a related failure cluster. P2 through P6 have not yet been started.",
    "Earlier controller launch attempts are classified as interrupted infrastructure/controller diagnostics, not successful evaluation results."
  ],
  "limitations": [
    "The current synthetic portfolio is a bounded decision harness, not evidence that Hiro can safely operate live calendar, email, task, reading-list, or web connectors.",
    "The first retry failure demonstrated that interruption state should be made even more explicit in the controller summary; the stable-variant fix addresses ledger identity, and future work should add per-partition resumability markers.",
    "Two of the three remaining misses expose arguable decision-policy ambiguity: quoted-history email handling and a clarification trigger without a clarification choice. Preserve these results and vary those dimensions in fresh cases rather than editing scored cases."
  ],
  "nextSteps": [
    "Advance to the next bounded fresh phase while increasing one structural dimension, retaining the repaired P1 as immutable baseline evidence.",
    "Release only the next bounded phase justified by novelty, completion, regression, and proposal evidence.",
    "Add a per-partition checkpoint if P1 evidence shows that a long partition needs safe mid-phase resumption.",
    "Keep all live integrations behind separate explicit authorization and production-specific evaluation."
  ],
  "disclosureNote": "This public entry omits credentials, held-out prompts, private data, sensitive runtime details, and actionable security weaknesses."
}
