{
  "schemaVersion": 2,
  "date": "2026.07.21",
  "publishedAt": "2026-07-21T14:47:12-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Manual proposal-only cycle completed; nightly evaluation reused by design",
  "publicationStatus": "Published",
  "executiveSummary": [
    "A user-requested manual run of Hiro's guarded proposal-only nightly workflow completed successfully after the normal readiness preflight confirmed that the local model and dashboard API were healthy.",
    "The workflow correctly reused the completed same-day public evaluation ledger record instead of creating a misleading second nightly evaluation. That existing record contains 16 observations, a 100% pass rate, and no evaluation failures.",
    "The manual cycle completed 22 lightweight regression probes: 20 passed and two deterministic date/time probes failed. It created two medium-risk, approval-required improvement proposals and four curiosity items, with no code change or automatic promotion.",
    "Because the evaluation was reused, this is not an additional valid completed night for readiness counting. The valid-night count remains unchanged; the next independent nightly evidence must come from a later calendar date."
  ],
  "workstreams": [
    {
      "title": "Readiness and execution",
      "status": "Completed",
      "details": [
        "Ran the guarded preflight before execution; the local model and dashboard API were ready.",
        "A first background attempt did not persist through the managed execution environment, so the same user-authorized workflow was relaunched in a persistent local context and completed.",
        "The completed proposal-only run finished in approximately 1.3 minutes and reported no unhandled run errors."
      ]
    },
    {
      "title": "Nightly public evaluation ledger",
      "status": "Reused; no new observations",
      "details": [
        "The public evaluation runner is intentionally idempotent for the same date, suite version, and source fingerprint.",
        "It reused the current valid record rather than appending duplicate evidence: 16 recorded observations, 100% pass rate, and zero failures.",
        "Held-out cases were not used, and no second valid-night count was claimed."
      ]
    },
    {
      "title": "Regression and improvement output",
      "status": "Completed; proposals pending approval",
      "details": [
        "Completed all 22 lightweight regression probes. Twenty passed; two deterministic date/time probes failed.",
        "Generated a medium-risk proposal to improve grounding before factual responses reach the final gate, backed by observed blocked responses.",
        "Generated a medium-risk proposal to evaluate a short-lived cache for recently verified factual answers, backed by an observed high-latency factual path.",
        "Both proposals require user approval and code changes; neither was applied."
      ]
    }
  ],
  "decisions": [
    "Treat the same-day evaluation reuse as expected idempotent behavior, not as a failed or a second successful nightly run.",
    "Keep the public evaluation's 16-observation, 100% result separate from the 22-probe regression diagnostic, because they exercise different evidence surfaces.",
    "Do not apply either generated proposal without explicit user approval, because both are medium-risk code changes.",
    "Do not inflate readiness counts with manual replays or infrastructure/tooling retries."
  ],
  "validation": [
    {
      "check": "Model and dashboard readiness preflight",
      "status": "passed",
      "result": "The local model and dashboard API were available before the workflow was launched."
    },
    {
      "check": "Manual proposal-only workflow",
      "status": "passed",
      "result": "The persistent run completed in approximately 1.3 minutes with no reported run errors."
    },
    {
      "check": "Nightly public evaluation",
      "status": "reused-valid-record",
      "result": "The same-day idempotent record was reused: 16 observations, 100% pass rate, and zero failures. No additional observations were executed."
    },
    {
      "check": "Regression probes",
      "status": "completed-with-findings",
      "result": "All 22 probes completed; 20 passed and two deterministic date/time probes failed."
    },
    {
      "check": "Automatic changes",
      "status": "not-applied",
      "result": "The workflow produced approval-required proposals only; no code, configuration, schedule, evaluation case, or model-setting change was applied."
    }
  ],
  "currentState": [
    "The valid completed evaluation for the current date remains the existing 16-observation, 100% passing public run.",
    "The manual proposal-only diagnostic cycle completed and its proposals remain pending user approval.",
    "Two deterministic date/time regression findings are available for investigation but were not treated as failures of the reused public evaluation.",
    "Readiness counting still requires another independent valid calendar-night evaluation before any process-expansion decision."
  ],
  "nextSteps": [
    "Observe the next scheduled nightly evaluation on a later calendar date and add it to readiness history only if it completes with valid evidence.",
    "Investigate the two deterministic date/time regression findings with bounded tests before considering a code change.",
    "Review the approval-required grounding and caching proposals separately; prioritize test diversity and adaptive testing before increasing repetitions.",
    "Continue excluding infrastructure failures, manual replays, and reused records from the successful-night count."
  ],
  "disclosureNote": "This public update intentionally omits user data, credentials, local paths, service addresses, raw evaluation content, and other operational deployment details."
}
