{
  "schemaVersion": 2,
  "date": "2026.07.23",
  "publishedAt": "2026-07-23T07:46:06-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Finite diversity campaign for Hiro Step 3",
  "publicationStatus": "Published",
  "executiveSummary": [
    "Designed a finite execution-grade roadmap Step 3 campaign that a lighter supervising model can initiate, monitor, resume, and close through deterministic controller commands and a fixed evidence decision table.",
    "The plan directly addresses the July 21 repetition failure. That session recorded 44,592 observations but only 2,820 structural signatures, a 93.68% structural-reuse rate. Seven of eight families had only four to eight effective topologies.",
    "The replacement campaign uses ten capability families, eight finite phases, a maximum of 360 planned cases including opening and closing canonical checks, global historical novelty rejection, family and topology caps, mastery retirement, sealed blind validation, and mandatory worker closeout."
  ],
  "workstreams": [
    {
      "title": "Repetition diagnosis",
      "status": "Completed",
      "details": [
        "Inspected the existing development-cycle runbook, v1 runner, generated catalogs, adaptive Daylab rotation, and stateful adversarial generator.",
        "Confirmed that the v1 runner extends beyond its nominal cycle count until the timebox and that signatures are checked only within each generated batch rather than against historical catalogs.",
        "Recomputed the continuous-session corpus: 929 cycles, 44,592 observations, 2,820 unique signatures, and 93.68% structural reuse."
      ]
    },
    {
      "title": "Finite capability portfolio",
      "status": "Designed",
      "details": [
        "Defined ten families spanning temporal calendar reasoning, task replanning, email grounding, cross-tool coordination, approval lifecycle, concurrency, recovery, memory continuity, source judgment, and productive autonomy.",
        "Defined a finite sequence from calibration through breadth, evolving state, cross-family composition, adaptive micro-waves, blind frontier, and closeout.",
        "Set the campaign target at 320-420 valid observations, with a manifest maximum of 348 noncanonical cases plus twelve opening and closing canonical workflow checks."
      ]
    },
    {
      "title": "Enforceable novelty and mastery",
      "status": "Designed",
      "details": [
        "Structural signatures exclude superficial names, prose, dates, and labels and instead encode constraint graphs, event ordering, tool classes, approval transitions, resource versions, fault sequences, temporal relations, oracle class, and family composition.",
        "The plan requires exact and near-duplicate rejection against repository suites and historical development and adversarial catalogs, at least 85% noncanary novelty, a fifteen-percent family cap, and a two-percent topology cap.",
        "Mastered family-band combinations retire after two fresh statistically supported waves and return only as small rotating canaries."
      ]
    },
    {
      "title": "Weaker-model supervision contract",
      "status": "Designed",
      "details": [
        "The supervisor follows machine-readable checkpoints and allowed-next-actions instead of reconstructing state from conversation history.",
        "A fixed table determines whether to increase difficulty, hold a band, create a bounded diagnostic wave, permit one evidence-based proposal, reject a batch, or close the campaign.",
        "Terminal handling covers verified duplicate workers, wrapper timeouts, endpoint failures, incomplete reports, exhausted generators, failed candidates, blind leakage, and hard-stop closeout."
      ]
    }
  ],
  "decisions": [
    "Do not use the v1 continuous development runner or repeated Daylab suites for roadmap Step 3.",
    "Make the deterministic controller responsible for variety, budgets, checkpointing, and stopping; the lighter model supervises rather than improvises the loop.",
    "Use a finite manifest and begin closeout thirty minutes before the deadline even when an earlier phase is incomplete.",
    "Permit at most one qualifying low- or medium-risk candidate per adaptive micro-wave and never alter evaluators, oracles, assertions, or failed cases to gain credit.",
    "If the required v2 controller cannot be implemented and validated within its bounded sixty-minute prerequisite, stop as step3_controller_not_ready rather than reverting to repetition."
  ],
  "validation": [
    {
      "check": "Historical structural-reuse audit",
      "status": "completed",
      "result": "The July 21 continuous session contained 44,592 cataloged cases and 2,820 unique structural signatures, yielding 93.68% reuse. Seven families had only four to eight signatures; long-context attention supplied 2,778."
    },
    {
      "check": "Plan structure validation",
      "status": "passed",
      "result": "The runbook contains all ten required families, all eight phases, the finite manifest, novelty/family/topology limits, mastery retirement, supervisor decision table, closeout contract, evidence packet, and lighter-model handoff prompt. Planned maximum is 360 cases including canonical open and close checks."
    },
    {
      "check": "Step 3 controller availability",
      "status": "not-ready",
      "result": "The specified hiro.benchmarks.development_cycle.step3 interface does not yet exist. The runbook therefore makes controller implementation and validation a bounded prerequisite and prohibits fallback to v1."
    },
    {
      "check": "Capability evaluation",
      "status": "not-run-by-design",
      "result": "No Step 3 evaluation campaign, model run, candidate change, production connector, service, or schedule was started during this planning session."
    }
  ],
  "currentState": [
    "The Step 3 execution plan is available in docs/HIRO_STEP3_DIVERSITY_CAMPAIGN.md.",
    "The existing development-cycle guide now points Step 3 operators to the finite campaign and explicitly disallows the v1 runner for this purpose.",
    "The plan is ready for controller implementation; it is not yet safe to start the long campaign with the old runner."
  ],
  "limitations": [
    "The v2 finite controller and expanded scenario generators remain to be implemented and tested.",
    "The ten-family portfolio specifies required structural dimensions but does not itself create their deterministic oracles.",
    "Historical signature comparison may require adapters for older catalogs and agent-harness reports with different schema versions.",
    "All planned evidence remains hermetic and does not establish permission or readiness for live connector writes."
  ],
  "nextSteps": [
    "Implement the finite Step 3 controller, expanded generator catalog, global novelty index, mastery state, and deterministic checkpoint interface.",
    "Test prepare, validate-plan, run/resume, status, and idempotent close commands, including duplicate-worker and deadline behavior.",
    "Run a short dry validation that generates and scores manifests without beginning the full campaign.",
    "After the readiness gate passes, hand the exact runbook prompt to the lighter supervising model for the finite campaign."
  ],
  "disclosureNote": "This public plan omits credentials, private held-out prompts, personal data, and actionable security weaknesses."
}
