{
  "schemaVersion": 2,
  "date": "2026.08.05",
  "publishedAt": "2026-08-05T17:49:26-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Adding durable quiet-hours alerts for Hiro's sandbox candidates",
  "publicationStatus": "Validated and published",
  "executiveSummary": [
    "Implemented proactive Telegram review alerts for eligible autonomous sandbox candidates that remain outside Hiro's automatic-promotion authority.",
    "Candidates discovered from 11:00 PM through 6:59 AM America/Los_Angeles are durably held until 7:00 AM Pacific; candidates found at other times are eligible for immediate delivery.",
    "The hold is a persistent SQLite outbox rather than an in-memory sleep, so pending alerts survive ordinary process restarts and are deduplicated by candidate identity.",
    "Hiro's improvement discovery remains evidence-driven: nightly deterministic evaluation failures, response-envelope error and latency patterns, repeated metacognitive failure clusters, and bounded curiosity hypotheses create specifications before sandbox construction begins."
  ],
  "workstreams": [
    {
      "title": "Pacific quiet-hours notification policy",
      "status": "Implemented and passed",
      "details": [
        "Added an explicit America/Los_Angeles delivery calculation using timezone-aware datetimes, including daylight-saving offsets.",
        "The quiet interval begins at 23:00 inclusive and ends at 07:00 exclusive. A candidate found during that interval receives a due timestamp at the next 07:00 Pacific boundary.",
        "Candidates found before 23:00 or at and after 07:00 are eligible for immediate notification.",
        "The notification includes candidate identity, opportunity, risk, hypothesis, bounded file scope, decision evidence, the absence of automatic promotion authority, and the local benchmark review location."
      ]
    },
    {
      "title": "Durable outbox and retries",
      "status": "Implemented and passed",
      "details": [
        "Added a local SQLite outbox under Hiro's runtime directory with one immutable candidate identity per notification record.",
        "Pending messages retain discovery time, due time, attempt count, last error, and sent time. Failed Telegram sends remain pending for a later scheduler retry.",
        "Re-queuing the same candidate cannot create a second outbox record or a routine duplicate notification.",
        "Ineligible or already promoted candidates never enter this review-alert outbox."
      ]
    },
    {
      "title": "Scheduler and sandbox integration",
      "status": "Implemented and passed",
      "details": [
        "After an eligible Stage 4 recommendation freezes, the autonomous sandbox queues its review notification and appends notification status to the experiment event trail.",
        "The long-running scheduler flushes due sandbox alerts every minute before checking whether nightly research itself is enabled.",
        "This ordering preserves timely 07:00 delivery even when the research schedule is temporarily disabled, provided the scheduler process remains running.",
        "Notification failure does not rewrite or invalidate the frozen candidate and evaluation packets."
      ]
    },
    {
      "title": "Improvement discovery map",
      "status": "Reviewed and documented in session evidence",
      "details": [
        "Nightly public evaluation harvests deterministic failed cases, groups them by capability category, and produces one bounded diagnostic specification for the highest-impact failure cluster.",
        "Recent response envelopes expose timeouts, errors, blocked answers, latency outliers, and validation outcomes that can become repeatable competence specifications.",
        "Metacognition records identify high-severity failures, repeated failure modes, weak-signal clusters, surprising successes worth generalizing, and missing instrumentation.",
        "Curiosity hypotheses can identify useful experiments, but diagnostic-only or causally incomplete ideas do not become code candidates until they satisfy the complete specification contract."
      ]
    }
  ],
  "decisions": [
    "Use a persistent outbox so quiet-hours delivery is not lost when the nightly evaluation process exits before 07:00.",
    "Define quiet hours in America/Los_Angeles rather than a fixed UTC offset so daylight-saving transitions remain correct.",
    "Alert only for eligible candidates waiting at the controlled promotion boundary; rejected candidates remain visible on the benchmark page without generating individual Telegram noise.",
    "Keep the benchmark dashboard as the canonical evidence record and make Telegram a concise review-needed signal only.",
    "Retain notification failures for retry instead of treating them as candidate evaluation failures or silently discarding them."
  ],
  "validation": [
    {
      "check": "Focused notification and integration suite",
      "status": "passed",
      "result": "18 tests passed in 1.81 seconds. Coverage includes the 22:59, 23:00, 06:59, and 07:00 Pacific boundaries, durable delayed delivery, deduplication, immediate daytime delivery, ineligible-candidate suppression, scheduler behavior, Telegram lifecycle boundaries, autonomous sandbox behavior, and dashboard APIs."
    },
    {
      "check": "Repository-wide regression suite",
      "status": "passed",
      "result": "323 tests passed in 126.37 seconds."
    },
    {
      "check": "Python syntax compilation",
      "status": "passed",
      "result": "The sandbox notification outbox, autonomous sandbox runner, and scheduler modules compiled successfully."
    },
    {
      "check": "Hiro journal generation and frontend build",
      "status": "passed",
      "result": "Timestamped-entry unit tests passed; the generator produced and validated 66 journal pages, and the TypeScript and Vite production build completed successfully."
    }
  ],
  "currentState": [
    "Eligible autonomous sandbox candidates now create a durable review alert after their frozen evaluation decision.",
    "Alerts discovered during the user-selected 23:00-07:00 Pacific quiet interval wait until 07:00 Pacific.",
    "The scheduler checks the outbox every minute and retries failed sends while retaining candidate-level deduplication.",
    "The benchmark page remains the full review surface for eligible, rejected, and incomplete autonomous candidates.",
    "No Telegram message or live candidate was generated during this implementation session."
  ],
  "limitations": [
    "Delivery requires Hiro's scheduler process to be running and its existing Telegram lifecycle marker, bot configuration, and allowed recipient configuration to be valid.",
    "If the scheduler is stopped at 07:00, a pending message sends on the first successful flush after it returns rather than being lost.",
    "The outbox provides durable candidate-level deduplication, but an operating-system failure after Telegram accepts a message and before the local sent update could theoretically produce one retry duplicate.",
    "Individual Telegram alerts are intentionally limited to eligible review candidates; rejected-candidate learning remains available through the benchmark dashboard.",
    "This work does not grant merge, deployment, restart, Stage 5 integration, or Stage 6 automatic-promotion authority."
  ],
  "nextSteps": [
    "Allow the first eligible autonomous sandbox candidate to exercise immediate or quiet-hours delivery and verify its dashboard record against the outbox event.",
    "Confirm the operational Telegram lifecycle and scheduler remain active through the 07:00 delivery window.",
    "Consider a single morning digest for rejected candidates only if dashboard review proves insufficient; avoid one alert per rejection.",
    "Continue tuning improvement discovery from fresh evaluation and runtime evidence without broadening the controlled promotion boundary."
  ],
  "disclosureNote": "This public entry contains no credentials, tokens, private held-out cases or expected answers, personal data, or actionable details about unresolved security weaknesses."
}
