{
  "schemaVersion": 2,
  "date": "2026.09.26",
  "publishedAt": "2026-09-26T16:03:00-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Comparing Nex and Empero model candidates for Hiro",
  "publicationStatus": "Comparison complete; no production replacement",
  "executiveSummary": [
    "Downloaded and hash-verified Nex-N2.5-mini and Empero Qwen3.8-35B-A3B Distill Q4_K_M artifacts, then compared them with the configured Qwen3.8-27B baseline.",
    "Completed 216 responses through the unchanged frozen Hiro model qualification suite. Nex was the strongest speed/assistant challenger, but neither candidate qualified as a general replacement."
  ],
  "workstreams": [
    {
      "title": "Reproducible model comparison",
      "status": "Completed",
      "details": [
        "Pinned Nex community and Empero publisher GGUF Q4_K_M artifacts by repository revision and SHA-256.",
        "Reused the existing 24-case suite, three repetitions per case, 8192 context and 4096 completion budget at temperature 0.1.",
        "Used an isolated loopback server and a separate ledger with existing benchmark code.",
        "Temporarily unloaded an unrelated LM Studio model with explicit authorization; restore after evaluation.",
        "Nex community artifact: ngquocvinh/Nex-N2.5-mini-GGUF at fe3093c81ec10e0388c5375468d2b6a86e0a6026.",
        "Empero publisher artifact: empero-ai/Qwen3.8-35B-A3B-Distill-GGUF at b1f9d1dcc3de8aa867669b0ab919384aeeb9b8d5.",
        "Both candidate downloads matched their published SHA-256 values; all three models passed runtime artifact checks.",
        "Initial baseline mmap startup was stopped before scoring after no visible progress. Successful comparisons used identical non-mapped loading."
      ]
    }
  ],
  "decisions": [
    "No production replacement or promotion is authorized by this comparison.",
    "Use identical loading and evaluation settings. Preserve loading failures separately from model quality results.",
    "Retain the configured model. Nex merits possible future constrained assistant evaluation because it matched the eight assistant scenarios with substantially lower latency, but this is not production qualification.",
    "Empero traded speed for lower scores on both lanes in this configuration.",
    "All three failed the automatic-deployment scenario; no real deployment occurred. Preserve benchmark failures without tuning prompts or relaxing gates."
  ],
  "validation": [
    {
      "check": "Frozen suite and comparability",
      "status": "passed",
      "result": "24 cases per model, three repetitions; suite/harness integrity and existing ledger comparability checks passed."
    },
    {
      "check": "Artifact verification",
      "status": "passed",
      "result": "Both downloaded GGUF hashes match publisher/repository LFS hashes. Baseline matches the existing pinned hash."
    },
    {
      "check": "baseline model qualification",
      "status": "rejected",
      "result": "Completed 72 responses: assistant 100.0%, self-improvement 86.4%, weighted 91.8%, fully passing 60/72. Median latency 3.97s; hard-gate failures 6."
    },
    {
      "check": "nex model qualification",
      "status": "rejected",
      "result": "Completed 72 responses: assistant 100.0%, self-improvement 82.2%, weighted 89.3%, fully passing 54/72. Median latency 0.57s; hard-gate failures 6."
    },
    {
      "check": "empero model qualification",
      "status": "rejected",
      "result": "Completed 72 responses: assistant 87.4%, self-improvement 82.7%, weighted 84.6%, fully passing 42/72. Median latency 1.19s; hard-gate failures 3."
    },
    {
      "check": "Session cleanup",
      "status": "passed",
      "result": "Previous LM Studio model restored with matching identity, context, parallelism and idle state; isolated benchmark server stopped."
    },
    {
      "check": "Journal validation",
      "status": "passed",
      "result": "npm run test:hiro and npm run build passed, including 243-entry generated validation."
    }
  ],
  "currentState": [
    "All requested models downloaded and tested; no production model/configuration changed.",
    "Both challengers fit text inference at 8K context, with peak sampled total GPU memory about 24.3 GiB. Larger contexts, vision and real long-horizon tool workflows were not tested.",
    "Fixed-seed repetitions are stability checks, not independent statistical trials. Public short scenarios cannot establish full production capability.",
    "Full local reports, raw responses, benchmark ledger, pinned profiles, server logs, sampled resources and plan retained."
  ],
  "nextSteps": [
    "Review results before authorizing any candidate activation or broader evaluation."
  ]
}
