{
  "title": "Lians LoCoMo judged benchmark results",
  "published": "2026-07-09",
  "updated": "2026-07-19",
  "benchmark": "LoCoMo",
  "questions": 1540,
  "conversations": 10,
  "metric": "LLM-judged QA accuracy",
  "answer_and_judge_harness": "mem0ai/memory-benchmarks",
  "results": [
    { "system": "Lians", "retrieval_cutoff": 10, "accuracy_percent": 83.4, "mean_tokens_per_question": 549, "full_context_share_percent": 3.0 },
    { "system": "Lians", "retrieval_cutoff": 20, "accuracy_percent": 87.3, "mean_tokens_per_question": 1083, "full_context_share_percent": 5.9 },
    { "system": "Lians", "retrieval_cutoff": 50, "accuracy_percent": 90.0, "mean_tokens_per_question": 2656, "full_context_share_percent": 14.6 },
    { "system": "Lians", "retrieval_cutoff": 200, "accuracy_percent": 92.9, "mean_tokens_per_question": 10283, "full_context_share_percent": 56.4 },
    { "system": "Mem0", "retrieval_cutoff": 200, "accuracy_percent": 92.5, "source": "Current result published in mem0ai/memory-benchmarks" }
  ],
  "caveat": "Both top-200 results use the mem0ai answer and judge pipeline, but the memory construction and retrieval settings differ. This is a comparison of reported results, not a controlled head-to-head experiment.",
  "methodology": "https://www.lians.ai/blog/locomo-benchmark",
  "source_report": "https://github.com/Lians-ai/Lians/blob/master/agentmem/docs/benchmarks/locomo-judged-2026-07-09.md",
  "harness": "https://github.com/mem0ai/memory-benchmarks"
}
