{
  "product": "Anamnesis",
  "publisher": "SMTRY (J. I. Ashley Consulting LLC)",
  "updated": "2026-08-24",
  "license": "CC BY 4.0 — cite anamnesis.smtry.ai/benchmarks",
  "protocol_summary": "Locked holdouts selected before any run; single-shot, no iteration against the test set; burned-question list for anything ever debugged; cross-family judges (Gemini + GPT) must agree; full prompts published.",
  "protocol_url": "https://anamnesis.smtry.ai/benchmarks/",
  "results": [
    {
      "id": "longmemeval_holdout2",
      "benchmark": "LongMemEval-S",
      "date": "2026-08-15",
      "configuration": "full pipeline with propositional sidecar (RFC-015)",
      "score": "23/30",
      "score_pct": 76.7,
      "holdout": "locked, fresh seed 211, single shot",
      "judges": "Gemini + GPT, exact agreement",
      "no_memory_baseline": "0/30 (Gemini judge) / 2/30 (GPT judge)",
      "caveat": "n=30 questions per holdout; small-sample variance applies"
    },
    {
      "id": "longmemeval_holdout1",
      "benchmark": "LongMemEval-S",
      "date": "2026-08-14",
      "configuration": "gist-only pipeline (pre-RFC-015)",
      "score": "22/30",
      "score_pct": 73.3,
      "holdout": "locked, seed 101, single shot",
      "judges": "Gemini + GPT",
      "no_memory_baseline": "2/30",
      "caveat": "S-scale haystacks (~115 sessions/question)"
    },
    {
      "id": "t1_fidelity",
      "benchmark": "T1 answer fidelity vs oracle",
      "date": "2026-08",
      "configuration": "distilled memory injection vs full source in context",
      "score": "94.2% vs 98.3% oracle",
      "token_discount": "~18x on the memory portion of a request",
      "caveat": "Discount compares memory injection to pasting full source; it is not a claim about total spend"
    },
    {
      "id": "local_tier_retrieval",
      "benchmark": "Local-tier distillation (bonsai-8b 2-bit, consumer laptop) vs hosted (Claude Haiku)",
      "date": "2026-08-18",
      "configuration": "identical pools, retrieval-stage metrics",
      "score": "retrieval parity 0.94",
      "cost": "$0 local vs ~$9.50 hosted per full-pool distillation run",
      "caveat": "Answering quality below hosted: 17/30 vs 22/30, temporal category 0/5 — local tier not yet temporal-safe"
    },
    {
      "id": "t5_search_tax",
      "benchmark": "T5 agent task efficiency, memory on vs off",
      "date": "2026-08",
      "score": "~6.8x fewer turns, ~5.7x faster, 3-10x cheaper",
      "caveat": "Holds only with authoritative provenance-backed injection; agents re-verify polite hints"
    }
  ]
}
