{
  "complete": false,
  "date_transport_review": {
    "adapter": "eval/public/adapters/longmemeval_session.py",
    "legacy_v2_replay": "Original undated behavior retained",
    "method": "Explicit session and question date text prefixes; question type retained only in evaluation metadata",
    "official_equivalence": false,
    "schema": "mnemetric.longmemeval-session/v3",
    "temporal_reasoning_measured": false,
    "tests": "tests/test_longmemeval_session_adapter.py"
  },
  "families": [
    {
      "benchmark": "LongMemEval",
      "implementation": {
        "adapter": "eval/public/adapters/longmemeval_session.py",
        "full_held_out_run_completed": false,
        "full_protocol_equivalence": false,
        "registered_adapter_uses_utility": true,
        "registered_development_suite": "longmemeval-session-development",
        "scoring_profile": "longmemeval-session-upstream-v1",
        "session_formula_utility": "eval/public/longmemeval_metrics.py",
        "synthetic_bundle_recomputation_tested": true,
        "tests": "tests/test_longmemeval_upstream_metrics.py",
        "uncertainty": "Mnemetric bootstrap intervals; not an upstream metric claim"
      },
      "protocol_requirements": [
        "Keep S, M and oracle variants separate.",
        "Abstention is identified by the _abs question-ID suffix; retrieval evaluation excludes those instances.",
        "QA correctness and turn/session retrieval metrics are different outcomes.",
        "Upstream recall_all requires every gold session; current historical Mnemetric recall_at_5 is fractional.",
        "Upstream DCG weights its first two ranks equally; historical Mnemetric nDCG uses the conventional discount.",
        "QA aggregation includes per-type, macro, overall and abstention accuracy; the script asserts judge gpt-4o-2024-08-06."
      ],
      "remaining_audit": [
        "QA judge prompts; full held-out normalization custody, timestamp preservation and turn-level evaluation",
        "Dataset revision custody",
        "Map every task to existing adapters and evidence"
      ],
      "repository": "https://github.com/xiaowu0162/LongMemEval",
      "review_scope": "README, retrieval formulas and metric aggregation reviewed; QA judge, full paper and dataset audit incomplete.",
      "reviewed_sources": [
        {
          "path": "README.md",
          "sha256": "c4ff45676683d9e2f7cf7d9099d26426f14635ec110dbb1da818d1019a142573",
          "url": "https://github.com/xiaowu0162/LongMemEval/blob/9e0b455f4ef0e2ab8f2e582289761153549043fc/README.md"
        },
        {
          "path": "src/evaluation/print_retrieval_metrics.py",
          "sha256": "58b70c0b562ea57372a7774a554c347cd908e901b77ac0149fc90b097b6f1b8f",
          "url": "https://github.com/xiaowu0162/LongMemEval/blob/9e0b455f4ef0e2ab8f2e582289761153549043fc/src/evaluation/print_retrieval_metrics.py"
        },
        {
          "path": "src/evaluation/print_qa_metrics.py",
          "sha256": "e9283933a0cefb7a0ded7365e436ae3d1be5aac41853325e6155d83bf07607f0",
          "url": "https://github.com/xiaowu0162/LongMemEval/blob/9e0b455f4ef0e2ab8f2e582289761153549043fc/src/evaluation/print_qa_metrics.py"
        },
        {
          "path": "src/retrieval/eval_utils.py",
          "sha256": "c98b8d1096877a15aa755c9de44fe33c195298466a2eb6f3c0f9f6bde8c72349",
          "url": "https://github.com/xiaowu0162/LongMemEval/blob/9e0b455f4ef0e2ab8f2e582289761153549043fc/src/retrieval/eval_utils.py"
        }
      ],
      "revision": "9e0b455f4ef0e2ab8f2e582289761153549043fc",
      "tasks": [
        {
          "classification": "question type",
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "question_type = single-session-user",
          "task": "single-session-user",
          "test_refs": []
        },
        {
          "classification": "question type",
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "question_type = single-session-assistant",
          "task": "single-session-assistant",
          "test_refs": []
        },
        {
          "classification": "question type",
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "question_type = single-session-preference",
          "task": "single-session-preference",
          "test_refs": []
        },
        {
          "classification": "question type",
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "question_type = temporal-reasoning",
          "task": "temporal-reasoning",
          "test_refs": []
        },
        {
          "classification": "question type",
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "question_type = knowledge-update",
          "task": "knowledge-update",
          "test_refs": []
        },
        {
          "classification": "question type",
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "question_type = multi-session",
          "task": "multi-session",
          "test_refs": []
        },
        {
          "classification": "cross-cutting overlay",
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "question_id suffix _abs",
          "task": "abstention",
          "test_refs": []
        }
      ]
    },
    {
      "benchmark": "LongMemEval-V2",
      "protocol_requirements": [
        "Keep web/enterprise domains and small/medium tiers identifiable.",
        "Memory accepts trajectories and returns text/image context under a token budget.",
        "Query receives question text and optional image, not gold labels or benchmark metadata.",
        "Accuracy, query latency and LAFS frontier gain require separate reporting."
      ],
      "remaining_audit": [
        "Exact scorer, LAFS reference frontier and aggregation audit",
        "Dataset/image custody",
        "Memory adapter implementation",
        "Fixed reader and resource validation"
      ],
      "repository": "https://github.com/xiaowu0162/LongMemEval-V2",
      "review_scope": "README task and protocol inventory; scorer, full paper and dataset audit incomplete.",
      "reviewed_sources": [
        {
          "path": "README.md",
          "sha256": "399cc61c9540e587017fccdde28b29b62ed66326425f16f08b5fe0457a593abd",
          "url": "https://github.com/xiaowu0162/LongMemEval-V2/blob/2cc8c540bdb87fe6761629b585e727e1c4704520/README.md"
        }
      ],
      "revision": "2cc8c540bdb87fe6761629b585e727e1c4704520",
      "tasks": [
        {
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "requires scorer/schema audit",
          "task": "Static state recall",
          "test_refs": []
        },
        {
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "requires scorer/schema audit",
          "task": "Dynamic state tracking",
          "test_refs": []
        },
        {
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "requires scorer/schema audit",
          "task": "Workflow knowledge",
          "test_refs": []
        },
        {
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "requires scorer/schema audit",
          "task": "Environment gotchas",
          "test_refs": []
        },
        {
          "coverage_status": "unmapped",
          "evidence_refs": [],
          "selector": "requires scorer/schema audit",
          "task": "Premise awareness",
          "test_refs": []
        }
      ]
    }
  ],
  "official_qa_setup_review": {
    "actual_judge_requests_executed": false,
    "compatibility_constraint": "httpx==0.27.2; upstream openai==1.35.1 fails with default httpx==0.28.1 during client initialization",
    "entrypoint": "eval/public/vendor/longmemeval/src/evaluation/evaluate_qa.py",
    "environment_lock": "eval/public/vendor/longmemeval/qa-compatibility.lock",
    "judge_model": "gpt-4o-2024-08-06",
    "label_parser": "substring yes in lowercased stripped response; unchanged upstream behavior",
    "output_suffix": ".eval-results-gpt-4o",
    "parameters": {
      "max_tokens": 10,
      "n": 1,
      "temperature": 0
    },
    "prompts": [
      "standard factual correctness",
      "temporal off-by-one tolerance",
      "knowledge-update allowance for old information alongside the required update",
      "preference rubric using personal information",
      "unanswerable-question recognition"
    ],
    "revision": "9e0b455f4ef0e2ab8f2e582289761153549043fc",
    "sdk_initialization_tested_without_network": true,
    "upstream_requirements": "eval/public/vendor/longmemeval/requirements-lite.txt"
  },
  "qa_aggregation_review": {
    "implementation": "eval/public/longmemeval_metrics.py:qa_metrics",
    "judge_executed": false,
    "judge_prompt_parity_verified": false,
    "registered_qa_adapter_uses_utility": false,
    "rules": [
      "Overall accuracy weights questions equally",
      "Task average weights all six question types equally",
      "Abstention is a cross-cutting subset selected by _abs in question ID",
      "Missing category or absent abstention subset is null instead of upstream NaN",
      "Exact reference/prediction ID sets and boolean labels are required"
    ],
    "source": "https://github.com/xiaowu0162/LongMemEval/blob/9e0b455f4ef0e2ab8f2e582289761153549043fc/src/evaluation/print_qa_metrics.py",
    "tests": "tests/test_longmemeval_qa_aggregation.py"
  },
  "reviewed_date": "2026-10-04",
  "schema": "mnemetric.upstream-task-inventory/v1"
}
