{
  "schemaVersion": 2,
  "date": "2026.09.27",
  "publishedAt": "2026-09-27T13:41:15-07:00",
  "timeZone": "America/Los_Angeles",
  "title": "Reviewing Contrastive Language Models as a local Hiro scorer",
  "publicationStatus": "Source review and deferred record; no integration or experiment",
  "executiveSummary": [
    "Reviewed CLM as a possible local ranking/decision component alongside the deferred Jev idea. It is a candidate scorer, not a replacement for Hiro's generative reasoning model or external discovery.",
    "The source documentation supports an exploratory evaluation, but does not establish GGUF equivalence, Hiro-specific calibration, or a production advantage."
  ],
  "workstreams": [
    {
      "title": "Primary-source and existing-state review",
      "status": "Completed",
      "details": [
        "Read https://github.com/Contrastive-LM/CLM, src/clm/embedder.py, src/clm/engine.py and https://huggingface.co/Contrastive-LM/CLM-v0.1-8B after following the supplied Reddit discussion.",
        "Reference heads use a frozen Qwen3-8B encoder with last-token pooling; the small head download does not include the encoder.",
        "Current serving documentation uses vLLM. The embedding client accepts an OpenAI-compatible endpoint, but that does not prove quantized llama.cpp embeddings preserve trained-head rankings or calibration.",
        "Inspected existing Hiro opportunity priority, continuous idea scoring and bounded external-lead selection. A learned scorer could be compared at these existing boundaries, without creating another research queue.",
        "Checked open handoffs; the existing ranked-autonomy handoff was not executed."
      ]
    },
    {
      "title": "Deferred catalog follow-up",
      "status": "Completed",
      "details": [
        "At the user's request, added deferred-clm-local-decision-scoring to the existing Research Map.",
        "Retained the supplied Reddit URL, upstream repository/model card/code links, review details, limitations, related Jev identity and reconsideration conditions.",
        "Updated the deferred-ideas navigation guide. No runtime or authority changes."
      ]
    }
  ],
  "decisions": [
    "Treat this as a potential local scoring companion, not evidence that Jev is obsolete.",
    "API-level similarities do not establish equal accuracy, calibration or generalization.",
    "Candidate-relative softmax scores are not calibrated probabilities that an action is safe or correct; all options may be poor.",
    "Retain existing independent evaluation, admission, promotion and authority checks. A ranking signal cannot authorize actions.",
    "A small future offline comparison should use frozen real Hiro decision records and independent labels, compare current rules, raw embedding similarity and CLM, and measure ranking quality, bad-option rejection, latency and memory. First test the required encoder/pooling behavior if using GGUF. No experiment is authorized by this review."
  ],
  "validation": [
    {
      "check": "Documentation and code review",
      "status": "completed",
      "result": "Read upstream implementation/model card and selected Hiro source functions; no runtime reproduction."
    },
    {
      "check": "Published benchmarks",
      "status": "not_reproduced",
      "result": "Reported agentic verifier scores rely on fine-tuned heads and small held-out task sets; they are not zero-shot Hiro results. Published latency hardware differs from the local RTX 5090."
    },
    {
      "check": "GGUF compatibility and local resources",
      "status": "not_run",
      "result": "No model downloaded, runtime installed, embedding parity tested, or coexistence memory measured."
    },
    {
      "check": "Deferred record",
      "status": "passed",
      "result": "Source schemas, unique identities, source references and existing Markdown export inclusion validated."
    }
  ],
  "currentState": [
    "CLM is a plausible research candidate. Existing Jev deferral and discovery configuration remain unchanged.",
    "Reference serving truncates at 2048 tokens unless configured otherwise. Longer-state behavior and quantization need evaluation.",
    "CLM ranks supplied options; it does not expand an absent discovery search space or independently produce novel options."
  ],
  "nextSteps": [
    "If selected for further work, evaluate one existing low-authority ranking decision offline before any integration."
  ]
}
