{
  "schemaVersion": 2,
  "date": "2026.07.21",
  "publishedAt": "2026-07-21T19:56:40-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "A harder overnight Hiro development cycle is designed",
  "publicationStatus": "Published",
  "executiveSummary": [
    "The repeated perfect scores on Hiro's current active suite rotation were diagnosed as evaluation saturation rather than proof that the broader assistant goals are solved.",
    "A reusable six-to-eight-hour Hiro development cycle was designed around fresh structurally generated cases, adaptive difficulty, blind validation, mastery retirement, and strict retain-or-revert gates.",
    "The plan targets 600 to 1,200 valid observations across eight capability families while limiting simple repetitions and keeping a meaningful moving frontier.",
    "The cycle authorizes only bounded low- and medium-risk candidate work under operator review; high-risk proposals and every consequential external action remain deferred for explicit approval."
  ],
  "workstreams": [
    {
      "title": "Saturation diagnosis",
      "status": "Completed",
      "details": [
        "Inspected the current adaptive rotation, Daylab orchestration, evaluation schema and scorer, self-improvement limits, safety policy, prior transition design, and the latest supervised-run evidence.",
        "Confirmed that the current lane rotates a small fixed suite pool and ordinarily caps a cycle at eight cases and sixteen observations.",
        "Concluded that repeated perfect results now primarily measure retention on known examples; the next useful budget should emphasize novel structures and harder capability combinations."
      ]
    },
    {
      "title": "Adaptive frontier curriculum",
      "status": "Designed",
      "details": [
        "Defined five difficulty bands from stable canaries through cross-domain frontier tasks, with advancement based on raw score and Wilson lower confidence bounds rather than a single perfect run.",
        "Defined eight task families covering compositional reasoning, constrained planning, simulated personal-assistant workflows, source judgment, recovery, long-context attention, calibration, and productive autonomy.",
        "Set a target frontier score of roughly sixty-five to eighty-five percent on fresh hard cases while retaining near-perfect essential canaries."
      ]
    },
    {
      "title": "Novelty and blind evidence",
      "status": "Designed",
      "details": [
        "Specified seeded structural generators, duplicate and near-duplicate rejection, deterministic oracles, immutable session case files, and provenance metadata for each admitted case.",
        "Split every macro-cycle into development, blind validation, and canary partitions, with batches pre-generated ahead of candidate work to reduce leakage and benchmark overfitting.",
        "Documented that blindness is procedural in the current single-host setup and that any accidentally exposed reserve case must be relabeled and replaced."
      ]
    },
    {
      "title": "Safe overnight operator protocol",
      "status": "Designed",
      "details": [
        "Defined eight to twelve diagnose, propose, review, implement, validate, and advance macro-cycles across a six-to-eight-hour timebox.",
        "Added measurable candidate retention gates, regression limits, invalid-run classification, infrastructure failure handling, stop conditions, checkpointing, and idempotent shutdown requirements.",
        "Provided a copy-ready handoff prompt for a lighter overnight operator and required a complete final evidence packet."
      ]
    }
  ],
  "decisions": [
    "Use Hiro development cycle as the durable name for the recurring supervised overnight process.",
    "Prefer broader test diversity and adaptive frontier movement over repeating mastered cases.",
    "Keep all evaluation traffic isolated from live tools, user data, networks, calendar, email, and production services; tonight's assistant workflows use only fictional embedded state.",
    "Do not modify the evaluation kernel, ledger, safety policy, authentication, services, schedules, or external integrations during the overnight loop.",
    "Require blind target improvement plus stable canaries before retaining a behavioral candidate, and preserve rejected results append-only.",
    "Treat the cycle as system improvement, not underlying model-weight training; any fine-tuning effort remains a separate reviewed project."
  ],
  "validation": [
    {
      "check": "Runbook consistency review",
      "status": "passed",
      "result": "The runbook contains the canonical loop, operating boundaries, observation and novelty targets, difficulty policy, task families, macro-cycle gates, timeboxed schedule, stop conditions, evidence packet, and operator handoff."
    },
    {
      "check": "Repository implementation",
      "status": "not-run-by-design",
      "result": "This session designed the overnight plan and did not start an evaluation cycle or implement a candidate."
    },
    {
      "check": "Journal tests and site build",
      "status": "passed",
      "result": "The required npm run test:hiro check passed, and the production npm run build generated and validated all journal pages before publication."
    }
  ],
  "currentState": [
    "The existing active suites are useful as stable canaries but are no longer sufficient as the main development frontier after repeated perfect runs.",
    "The durable overnight runbook is written in Hiro's repository and is ready to hand to a lighter operator after publication validation.",
    "No new evaluation run, service change, schedule change, candidate implementation, or external action occurred during this design session."
  ],
  "nextSteps": [
    "Launch the named Hiro development cycle with the runbook's handoff prompt and a configured six-to-eight-hour hard stop.",
    "Have the overnight operator implement and test the minimal session runner if it is absent, stopping instead of reverting to mastered-suite repetition if that cannot be done safely within forty-five minutes.",
    "Review the morning evidence packet, especially blind frontier gains, retained versus rejected changes, invalid runs, and the deferred high-risk queue.",
    "Use repeated overnight evidence to decide when a separately approved hermetic tool-state harness or weight-training project is warranted."
  ],
  "disclosureNote": "This public design update omits credentials, private held-out prompts, personal data, local deployment details, and actionable security information."
}
