{
  "benchmarks": [
    {
      "adapter": "longmemeval_session.py",
      "elements": [
        {
          "capabilities": [
            "C03"
          ],
          "element": "Session evidence retrieval",
          "modules": [
            "M02"
          ]
        },
        {
          "capabilities": [
            "C04"
          ],
          "element": "Temporal questions",
          "modules": [
            "M03"
          ]
        },
        {
          "capabilities": [
            "C05"
          ],
          "element": "Knowledge updates",
          "modules": [
            "M04"
          ]
        },
        {
          "capabilities": [
            "C12"
          ],
          "element": "Abstention",
          "modules": [
            "M10"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B01",
      "interpretation_limit": "QA accuracy and retrieval quality answer different questions. Neither alone establishes deletion, authorization or recovery. Our older retrieval scores use different formulas; the new session adapter has only development validation.",
      "name": "LongMemEval",
      "purpose": "Long-conversation question answering: extracting facts, combining sessions, tracking updates, temporal reasoning and abstaining. Retrieval evaluation separately measures whether supporting sessions or turns were found.",
      "source": "https://github.com/xiaowu0162/LongMemEval",
      "system_target": "Mnemosyne development integration",
      "test": "test_longmemeval_session_adapter.py"
    },
    {
      "adapter": null,
      "elements": [
        {
          "capabilities": [
            "C03",
            "C04",
            "C05"
          ],
          "element": "State recall and change",
          "modules": [
            "M02",
            "M03",
            "M04"
          ]
        },
        {
          "capabilities": [
            "C07",
            "C23"
          ],
          "element": "Workflow and environment knowledge",
          "modules": [
            "M14"
          ]
        },
        {
          "capabilities": [
            "C12"
          ],
          "element": "Premise awareness",
          "modules": [
            "M10"
          ]
        },
        {
          "capabilities": [
            "C21"
          ],
          "element": "Visual trajectory context",
          "modules": [
            "M19"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B02",
      "interpretation_limit": "Trajectory QA is not proof of successful action execution. Image input support, timestamps and trajectory semantics must survive adaptation; a text-only adapter cannot claim full multimodal coverage.",
      "name": "LongMemEval-V2",
      "purpose": "Evaluates static state recall, dynamic state tracking, workflow knowledge, environment gotchas and premise awareness using agent trajectories, including visual context.",
      "source": "https://github.com/xiaowu0162/LongMemEval-V2",
      "system_target": "No implemented system integration claimed",
      "test": null
    },
    {
      "adapter": "hipporag_multihop.py",
      "elements": [
        {
          "capabilities": [
            "C03",
            "C23"
          ],
          "element": "Multi-hop evidence retrieval and answering",
          "modules": [
            "M02",
            "M14"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B03",
      "interpretation_limit": "A graph-based reference system is not itself a universal memory benchmark. Reader quality and retrieval both affect QA; graph benefit requires a controlled ablation.",
      "name": "HippoRAG evaluation datasets",
      "purpose": "HippoRAG is a retrieval system and framework. MuSiQue, 2WikiMultiHopQA and HotpotQA supply evidence-retrieval and multi-hop question-answering tasks.",
      "source": "https://github.com/OSU-NLP-Group/HippoRAG",
      "system_target": "Mnemosyne development integration",
      "test": "test_public_hipporag.py"
    },
    {
      "adapter": "memoryagentbench.py",
      "elements": [
        {
          "capabilities": [
            "C03"
          ],
          "element": "Retrieval and long-range understanding",
          "modules": [
            "M02"
          ]
        },
        {
          "capabilities": [
            "C08",
            "C23"
          ],
          "element": "Learning from experience",
          "modules": [
            "M06",
            "M14"
          ]
        },
        {
          "capabilities": [
            "C05"
          ],
          "element": "Conflict resolution",
          "modules": [
            "M04"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B04",
      "interpretation_limit": "A development adapter does not reproduce the complete upstream task distribution or demonstrate durable storage, access control or operational recovery.",
      "name": "MemoryAgentBench",
      "purpose": "Incremental interaction tests accurate retrieval, test-time learning, long-range understanding and conflict resolution.",
      "source": "https://github.com/HUST-AI-HYZ/MemoryAgentBench",
      "system_target": "Mnemosyne development integration",
      "test": "test_public_memoryagentbench.py"
    },
    {
      "adapter": "beam.py",
      "elements": [
        {
          "capabilities": [
            "C03",
            "C04",
            "C05"
          ],
          "element": "Long-conversation memory",
          "modules": [
            "M02",
            "M03",
            "M04"
          ]
        },
        {
          "capabilities": [
            "C24"
          ],
          "element": "Scale and resource accounting",
          "modules": [
            "M01",
            "M02"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B05",
      "interpretation_limit": "A small local fixture cannot establish million-token scalability. Context length, ingest time, query latency, model and resource costs must accompany quality scores.",
      "name": "BEAM",
      "purpose": "Measures memory abilities over long, coherent conversations, including million-token-scale settings.",
      "source": "https://github.com/mohammadtavakoli78/BEAM",
      "system_target": "Mnemosyne development integration",
      "test": "test_public_beam.py"
    },
    {
      "adapter": "locomo.py",
      "elements": [
        {
          "capabilities": [
            "C03",
            "C04"
          ],
          "element": "Conversational QA and temporal events",
          "modules": [
            "M02",
            "M03"
          ]
        },
        {
          "capabilities": [
            "C08"
          ],
          "element": "Event summarization",
          "modules": [
            "M06"
          ]
        },
        {
          "capabilities": [
            "C21"
          ],
          "element": "Multimodal dialogue",
          "modules": [
            "M19"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B06",
      "interpretation_limit": "Our adapter and replay components do not imply coverage of every upstream task. QA results alone cannot establish summarization or multimodal generation performance.",
      "name": "LoCoMo",
      "purpose": "Long-term conversational memory evaluation includes question answering, event summarization and multimodal dialogue generation.",
      "source": "https://github.com/snap-research/locomo",
      "system_target": "Mnemosyne development integration",
      "test": null
    },
    {
      "adapter": null,
      "elements": [
        {
          "capabilities": [
            "C03"
          ],
          "element": "Remembering and reasoning",
          "modules": [
            "M02"
          ]
        },
        {
          "capabilities": [
            "C05",
            "C10"
          ],
          "element": "Updated or deleted information in answers",
          "modules": [
            "M04",
            "M08"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B07",
      "interpretation_limit": "Not producing deleted information in an answer does not prove its physical removal from stores, indexes or backups. Those require separate residue checks.",
      "name": "Memora / FAMA",
      "purpose": "Memora evaluates conversation-grounded remembering and forgetting. FAMA is a combined evaluation metric, not a second independent benchmark.",
      "source": "https://github.com/geniesinc/Memora",
      "system_target": "No implemented system integration claimed",
      "test": null
    },
    {
      "adapter": null,
      "elements": [
        {
          "capabilities": [
            "C08",
            "C23"
          ],
          "element": "Experience reuse in multi-session action",
          "modules": [
            "M06",
            "M14"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B08",
      "interpretation_limit": "Task success reflects the agent, tools and memory together. Isolating memory contribution requires matched agents and memory ablations.",
      "name": "MemoryArena",
      "purpose": "Interdependent, multi-session agent tasks test whether prior actions and feedback help later work, including navigation, planning, search and reasoning.",
      "source": "https://arxiv.org/abs/2602.16313",
      "system_target": "No implemented system integration claimed",
      "test": null
    },
    {
      "adapter": null,
      "elements": [
        {
          "capabilities": [
            "C07",
            "C08",
            "C23"
          ],
          "element": "Procedural learning and transfer",
          "modules": [
            "M06",
            "M14"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B09",
      "interpretation_limit": "Transfer across tested roles does not establish universal transfer or forgetting. This is a procedural-memory evaluation, not a physical-erasure test.",
      "name": "AFTER",
      "purpose": "Evaluates procedural skill revision, specialization, reuse and transfer across tasks, roles and model backbones.",
      "source": "https://github.com/DavydenkoGr/AFTER",
      "system_target": "No implemented system integration claimed",
      "test": null
    },
    {
      "adapter": null,
      "elements": [
        {
          "capabilities": [
            "C08",
            "C23"
          ],
          "element": "Workflow execution and learning",
          "modules": [
            "M06",
            "M14"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B10",
      "interpretation_limit": "Enterprise task success is an end-to-end outcome; a matched configuration is needed to attribute gains specifically to memory.",
      "name": "STATE-Bench",
      "purpose": "Benchmarks agents on enterprise workflows, including learning through memory, skills and prompt optimization.",
      "source": "https://github.com/microsoft/STATE-Bench",
      "system_target": "No implemented system integration claimed",
      "test": null
    },
    {
      "adapter": null,
      "elements": [
        {
          "capabilities": [
            "C03",
            "C07"
          ],
          "element": "Speaker and audience-grounded retrieval",
          "modules": [
            "M02"
          ]
        },
        {
          "capabilities": [
            "C04",
            "C05"
          ],
          "element": "Updates and temporal reasoning",
          "modules": [
            "M03",
            "M04"
          ]
        },
        {
          "capabilities": [
            "C12"
          ],
          "element": "Abstention",
          "modules": [
            "M10"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B11",
      "interpretation_limit": "Remembering which person said something differs from enforcing access permissions. Multi-party QA alone is not authorization or tenant-isolation proof.",
      "name": "GroupMemBench",
      "purpose": "Multi-party conversations test group dynamics, speaker-grounded beliefs and audience-specific language. Queries include multi-hop reasoning, updates, ambiguity, user-implicit reasoning, time and abstention.",
      "source": "https://arxiv.org/abs/2605.14498",
      "system_target": "No implemented system integration claimed",
      "test": null
    },
    {
      "adapter": null,
      "elements": [
        {
          "capabilities": [
            "C03",
            "C16"
          ],
          "element": "Authorization-aware recall",
          "modules": [
            "M02",
            "M11"
          ]
        },
        {
          "capabilities": [
            "C05",
            "C10"
          ],
          "element": "Updates and active forgetting",
          "modules": [
            "M04",
            "M08"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B11",
      "interpretation_limit": "Answer-level leakage and forgetting checks do not establish physical erasure. Our local signed-identity checks are related engineering checks, not GateMem execution.",
      "name": "GateMem",
      "purpose": "Shared memory across multiple principals tests useful recall under updates, contextual authorization boundaries and active forgetting after deletion.",
      "source": "https://arxiv.org/abs/2606.18829",
      "system_target": "No implemented system integration claimed",
      "test": null
    },
    {
      "adapter": "pm_bench_triggerbench.py",
      "elements": [
        {
          "capabilities": [
            "C13"
          ],
          "element": "Delayed intentions and triggering",
          "modules": [
            "M12"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B12",
      "interpretation_limit": "Recognizing an intention is not executing it at the correct moment. Local authored action probes do not constitute results on the upstream benchmark.",
      "name": "PM-Bench",
      "purpose": "Prospective-memory evaluation tests remembering delayed intentions and acting on future cues or state changes while other activity continues.",
      "source": "https://arxiv.org/abs/2607.12385",
      "system_target": "Mnemosyne development integration",
      "test": "test_public_pm_bench_triggerbench.py"
    },
    {
      "adapter": "pm_bench_triggerbench.py",
      "elements": [
        {
          "capabilities": [
            "C13"
          ],
          "element": "Cue-dependent future action",
          "modules": [
            "M12"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B12",
      "interpretation_limit": "A repository-specific probe shares a construct but not necessarily upstream tasks, scoring or difficulty. Official comparability remains unestablished.",
      "name": "TriggerBench",
      "purpose": "Tests prospective memory in assistant and professional workflows: retaining an intention until its triggering circumstances arrive.",
      "source": "https://arxiv.org/abs/2606.23459",
      "system_target": "Mnemosyne development integration",
      "test": "test_public_pm_bench_triggerbench.py"
    },
    {
      "adapter": null,
      "elements": [
        {
          "capabilities": [
            "C08",
            "C23"
          ],
          "element": "Knowledge and execution experience reuse",
          "modules": [
            "M06",
            "M14"
          ]
        }
      ],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B13",
      "interpretation_limit": "Improvement in a particular agent loop does not prove model-independent learning or causal memory benefit without controls.",
      "name": "EvoMemBench",
      "purpose": "Evaluates self-evolving agent memory across within-episode and cross-episode use, spanning knowledge and execution content.",
      "source": "https://arxiv.org/abs/2605.18421",
      "system_target": "No implemented system integration claimed",
      "test": null
    },
    {
      "adapter": null,
      "elements": [],
      "equivalence": "Construct overlap only; official protocol equivalence not established",
      "id": "B14",
      "interpretation_limit": "The primary paper could not be fully retrieved in this review. We leave the mapping unassigned rather than infer visual, game or operational coverage from its name.",
      "name": "EMemBench",
      "purpose": "Registered in our slate as an interactive memory benchmark. Detailed task-level mapping is pending source verification.",
      "source": "https://openreview.net/pdf?id=zzndQeR4Ay",
      "system_target": "No implemented system integration claimed",
      "test": null
    }
  ],
  "checks": [
    {
      "benchmarks": [
        "B01"
      ],
      "modules": [
        "M02"
      ],
      "name": "Session metric formulas",
      "remaining_gap": "Full dataset, turn-level metrics and official QA judge remain separate.",
      "scope": "Matches audited upstream session discount and recall definitions at 5 and 10; checks bundle recomputation.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_longmemeval_upstream_metrics.py"
    },
    {
      "benchmarks": [
        "B01"
      ],
      "modules": [
        "M02"
      ],
      "name": "Session adapter protocol",
      "remaining_gap": "Excluding abstention questions from retrieval does not test QA abstention. Preserving date text does not establish temporal reasoning, bitemporal storage or full official execution.",
      "scope": "Synthetic normalization, explicit retrieval abstention exclusion, ten-hit retention, empty retrieval, date-prefix transport and real CLI bundle.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_longmemeval_session_adapter.py"
    },
    {
      "benchmarks": [
        "B03"
      ],
      "modules": [
        "M02",
        "M14"
      ],
      "name": "Multi-hop adapter",
      "remaining_gap": "Not a controlled comparison with HippoRAG or proof of graph benefit.",
      "scope": "Development adapter contract for multi-hop retrieval and answering.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_public_hipporag.py"
    },
    {
      "benchmarks": [
        "B04"
      ],
      "modules": [
        "M02",
        "M04",
        "M06"
      ],
      "name": "Incremental memory adapter",
      "remaining_gap": "Full upstream categories and official scoring need protocol audit and retained runs.",
      "scope": "Development adapter contract.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_public_memoryagentbench.py"
    },
    {
      "benchmarks": [
        "B05"
      ],
      "modules": [
        "M02"
      ],
      "name": "Long-context adapter",
      "remaining_gap": "Small fixtures do not establish 1M/10M-scale performance.",
      "scope": "Development adapter contract.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_public_beam.py"
    },
    {
      "benchmarks": [
        "B12"
      ],
      "modules": [
        "M12"
      ],
      "name": "Prospective action probes",
      "remaining_gap": "Not official PM-Bench or TriggerBench tasks or scores.",
      "scope": "Authored development cases related to future-action constructs.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_public_pm_bench_triggerbench.py"
    },
    {
      "benchmarks": [
        "B01"
      ],
      "modules": [
        "M10",
        "M13"
      ],
      "name": "Working-memory selection and abstention",
      "remaining_gap": "Not semantic QA abstention, learned relevance or full working-memory capacity.",
      "scope": "Related construct only: explicit relevance and deterministic policy checks.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_public_working_memory_action_probe.py"
    },
    {
      "benchmarks": [
        "B11"
      ],
      "modules": [
        "M11",
        "M13"
      ],
      "name": "Scope identity and TTL expiry",
      "remaining_gap": "Not GroupMemBench group reasoning or GateMem workload execution. TTL is not temporal-question reasoning.",
      "scope": "GateMem-related authorization boundary checks; TTL and scoped expiry are independent engineering contracts.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_working_memory_cli_contract.py"
    },
    {
      "benchmarks": [],
      "modules": [
        "M16",
        "M18"
      ],
      "name": "CLI/MCP declarations",
      "remaining_gap": "Command counts are not behavior coverage; MCP inventory is separately available below.",
      "scope": "Static interface inventory helps locate integration surfaces.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_cli_coverage_inventory.py"
    },
    {
      "benchmarks": [],
      "modules": [
        "M16",
        "M18"
      ],
      "name": "MCP declarations",
      "remaining_gap": "Does not establish transport parity or successful tool execution.",
      "scope": "Static tool inventory and signature correspondence.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_mcp_coverage_inventory.py"
    },
    {
      "benchmarks": [],
      "modules": [
        "M16"
      ],
      "name": "Backend/transport parity",
      "remaining_gap": "Full backend, SDK and remote transport matrix remains open.",
      "scope": "Bounded local parity contract; no direct upstream task equivalence claimed.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_public_wmbs_m16.py"
    },
    {
      "benchmarks": [],
      "modules": [
        "M20"
      ],
      "name": "Result publication contracts",
      "remaining_gap": "Rendering checks say nothing about memory-system quality.",
      "scope": "Checks evidence presentation and publication behavior.",
      "status": "Development check; not an official benchmark result",
      "system": "Mnemosyne integration or evaluator contract; no competitor execution implied",
      "test": "tests/test_leaderboard_render.py"
    }
  ],
  "complete": false,
  "reviewed": "2026-10-04",
  "schema": "mnemetric.benchmark-crosswalk/v1",
  "scope": "Family-level construct map plus selected reviewed development checks. Not an exhaustive per-task or per-test audit."
}
