{
  "schemaVersion": 2,
  "date": "2026.08.11",
  "publishedAt": "2026-08-11T20:38:02-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "A rotating everyday-question audit now feeds Hiro's improvement queue",
  "publicationStatus": "Implemented, activated, and producing live queue items",
  "executiveSummary": [
    "Hiro now tests its everyday conversational behavior proactively instead of relying only on user-reported failures and a small set of known failure patterns.",
    "A versioned suite contains 32 ordinary questions across 16 categories, including preferences, explanations, arithmetic, practical instructions, writing, summarization, recommendations, trip planning, directions, current information, follow-up continuity, ambiguity, action safety, correction recovery, and prompt-injection resistance.",
    "Four cases run every fifteen minutes after a quiet production window. The complete suite rotates in approximately two hours without calling the production chat endpoint, reading user memory, accessing personal accounts, or performing external actions.",
    "Every response passes through Hiro's real universal response boundary and a deterministic task contract. A failed case becomes a deduplicated lane-two item in the same ranked continuous-improvement queue used by other reliability findings.",
    "The first live scheduled batch passed the ordinary comparison and general-knowledge cases and found two additional usability failures. Both entered the live queue automatically; one immediately advanced to candidate construction.",
    "The full repository suite passed all 557 tests, Hiro restarted successfully through the hidden launcher, and the Observatory now displays audit coverage and the latest batch result."
  ],
  "workstreams": [
    {
      "title": "Broad everyday interaction suite",
      "status": "Activated",
      "details": [
        "Added two public cases in each of sixteen categories for balanced initial coverage rather than concentrating on one resolver or assistant niche.",
        "Cases test usefulness properties such as relevance, requested structure, necessary detail, concision, continuity, clarification, evidence use, safe action boundaries, and resistance to instructions embedded in source text.",
        "The case order is category-interleaved, so each four-question batch covers different behavior types before the suite repeats.",
        "The suite is versioned and included in Hiro's code fingerprint, allowing failures and later repairs to be tied to an exact evaluation definition and code revision."
      ]
    },
    {
      "title": "Isolated live-model execution",
      "status": "Activated",
      "details": [
        "Each case uses Hiro's configured local evaluation model and then passes the answer through the same universal final gate used for user-facing responses.",
        "Tool-shaped cases use synthetic evidence receipts and synthetic resolvers. They verify grounding, routing expectations, and answer construction without web calls, account access, or side effects.",
        "Execution waits until production has been quiet, runs no more than four cases per batch, and enforces a thirty-second timeout per case.",
        "Audit results record category, failures, latency, validator outcome, and prompt-safe status while keeping captured replay material out of the public dashboard."
      ]
    },
    {
      "title": "Automatic queue handoff",
      "status": "Activated",
      "details": [
        "Every failed contract is converted into a lane-two automatic-quality incident with a stable failure signature, affected task category, intended behavior, expected metrics, and private replay evidence.",
        "Duplicate failures from the same revision do not create duplicate queue items.",
        "The existing continuous engine reproduces an audit failure against the current response boundary before candidate construction, so a repair that already removed the failure can close it without unnecessary mutation.",
        "Candidate work retains the existing isolated worktree, targeted tests, full-suite checks, eight-hour canary, and stable-governor promotion path."
      ]
    },
    {
      "title": "Live failure signal repair",
      "status": "Activated",
      "details": [
        "Production answers blocked by the universal validator now become repair-needed evidence instead of remaining informational metadata.",
        "Synthetic and evaluation calls omit the production marker, preventing test traffic from entering the live incident harvester through this path.",
        "The grounding classifier no longer treats the words 'which', 'check', or a bare 'trip' as unconditional evidence requirements.",
        "Ordinary preference comparisons, short conversational follow-ups, and packing-list requests can now receive direct answers while genuinely current queries still require grounding."
      ]
    },
    {
      "title": "Observatory coverage visibility",
      "status": "Activated",
      "details": [
        "The current-improvement page now reports the question-suite size, category count, and latest batch pass/fail result.",
        "Audit-generated failures appear as ordinary ranked idea cards with their category, scores, affected implementation surface, tests, and eventual outcome.",
        "The read-only API exposes prompt-safe audit status without exposing case prompts, model responses, private replay text, or user data."
      ]
    }
  ],
  "decisions": [
    "Extend the existing continuous loop instead of reviving DayLab or NightLab as a competing improvement system.",
    "Test a broad ordinary-assistant surface continuously, not only the failure types already encountered by the user.",
    "Run a small frequent batch after production quiet time so the whole initial suite is covered in about two hours without sustained model contention.",
    "Use the local model plus the real universal response boundary for behavioral realism, while substituting synthetic evidence for live tools and personal accounts.",
    "Require deterministic contracts rather than asking a model to grade its own answers.",
    "Queue audit failures as reliability work automatically, with no routine user authorization step and no bypass of candidate tests or probation.",
    "Treat a finite suite as a growing sample of behavior rather than a claim that every possible question is covered."
  ],
  "validation": [
    {
      "check": "Focused interaction, scheduler, queue, and dashboard suites",
      "status": "passed",
      "result": "37 focused tests passed after the final audit, grounding, queue-handoff, and Observatory changes."
    },
    {
      "check": "Full repository suite",
      "status": "passed",
      "result": "557 tests passed in 160.21 seconds."
    },
    {
      "check": "Compilation and diff validation",
      "status": "passed",
      "result": "Python compilation and Git whitespace validation completed successfully."
    },
    {
      "check": "First isolated local-model batch",
      "status": "passed with findings",
      "result": "Four categories ran in 22.7 seconds. Preference comparison and general knowledge passed; concept-explanation concision and arithmetic explanation failed their contracts and were converted into isolated queue records."
    },
    {
      "check": "First live scheduled batch",
      "status": "passed with findings",
      "result": "The enabled scheduler independently produced the same two failures, created two live lane-two queue items, and advanced the higher-ranked item to candidate construction."
    },
    {
      "check": "Live service and Observatory",
      "status": "passed",
      "result": "Hiro restarted through the detached hidden launcher, reported healthy with the local model connected, and served the Question audit and Everyday audit status on the Observatory."
    },
    {
      "check": "Notification state",
      "status": "passed",
      "result": "The Telegram enablement marker remains absent."
    }
  ],
  "currentState": [
    "Hiro is running revision db9c924 through the detached hidden launcher.",
    "The everyday audit is enabled at four cases every fifteen minutes after a production quiet window.",
    "The suite currently contains 32 questions across 16 categories and completes one rotation in approximately two hours when production remains available.",
    "Two audit-discovered failures are active in the live ranked queue: one at candidate construction and one awaiting local reproduction.",
    "Ordinary preference questions no longer fail merely because they contain the word 'which'.",
    "Future unexpected production validator blocks are marked for automatic repair intake.",
    "Telegram notifications remain disabled."
  ],
  "limitations": [
    "Thirty-two cases cannot represent every possible user question. Coverage must grow from new real-world failure classes, category-level blind spots, and repeated weak scores.",
    "The audit uses synthetic tool evidence and does not exercise live personal-account providers. Separate provider integration tests remain necessary for Gmail, calendar, and other account-backed behavior.",
    "Deterministic contracts can measure concrete usefulness properties but cannot perfectly judge taste, nuance, humor, or every subjective preference.",
    "The first suite has two examples per category. Statistical confidence will improve as the corpus expands and cases receive repeated observations.",
    "A queued failure is evidence that repair work started, not a guarantee that its first autonomous candidate will pass. Candidate rejection and another bounded attempt remain valid outcomes."
  ],
  "nextSteps": [
    "Allow the first two-hour rotation to complete and review category-level pass rates rather than only total pass rate.",
    "Follow the two initial findings through candidate construction, targeted tests, probation, and promotion or rejection.",
    "Add new cases when real conversations reveal a behavior not represented by the current sixteen categories.",
    "Introduce carefully isolated provider-contract cases for directions, calendar, email, and other personal tools without using live user accounts.",
    "Track recurring failure clusters and increase their case density until the weak category stabilizes."
  ],
  "disclosureNote": "This public entry contains no credentials, private user text, private session identifiers, raw captured replay content, hidden reasoning, or actionable unresolved security details."
}
