{
  "scope": "Single preselected feasibility case, not a comparative benchmark result",
  "case_id": "factconsolidation_sh_6k:0:0",
  "references_sha256": "619f58549e9347a1f50b2e35a03f29a29e94de71f1b6e23da98aa7d4b01cb29e",
  "source_response_hashes": {
    "mnemosyne": "2a167e823f2bc67ea3637c9cb58058ecde414d46e614faad83c04476b386cfe4",
    "bm25": "1aca0543732156b86206840b78764e28efb35d5433b766590512185cad090eda"
  },
  "reports": {
    "mnemosyne": {
      "sub_dataset": "factconsolidation_sh_6k",
      "task_config": {
        "path": "configs/data_conf/Conflict_Resolution/Factconsolidation_sh_6k.yaml",
        "sha256": "814bf2eca1d07018262a860819db8fc7ddcb7d1a866839caec53ef2993c625e6",
        "execution_scope": "Postprocessing selection only; full task setup not executed."
      },
      "primary_metric": "substring_exact_match",
      "official_judge_required": false,
      "schema": "mnemetric.memoryagentbench-answer-metrics/v1",
      "official_full_benchmark_run": false,
      "scope": "Answer scoring only; task-specific upstream postprocessing when sub_dataset is set. No agent execution or judge scoring. Summary and LongMemEval lexical metrics are auxiliary and do not replace their official LLM-judge evaluations.",
      "upstream_revision": "538026089d1a8a8eff05121d0db89b388f360eba",
      "source_sha256": "d77976be409298970614d477a9d8003850caddb0510e56a7e821a037d98493a2",
      "python": "3.12.14",
      "dependencies": {
        "numpy": "1.26.4",
        "nltk": "3.9.1",
        "tiktoken": "0.7.0",
        "rouge-score": "0.1.2",
        "editdistance": "0.8.1"
      },
      "cases": [
        {
          "case_id": "factconsolidation_sh_6k:0:0",
          "metrics": {
            "exact_match": false,
            "f1": 0,
            "substring_exact_match": false,
            "rougeL_f1": 0.0,
            "rougeL_recall": 0.0,
            "rougeLsum_f1": 0.0,
            "rougeLsum_recall": 0.0
          },
          "postprocessing": {
            "parsed_output": "association football"
          }
        }
      ]
    },
    "bm25": {
      "sub_dataset": "factconsolidation_sh_6k",
      "task_config": {
        "path": "configs/data_conf/Conflict_Resolution/Factconsolidation_sh_6k.yaml",
        "sha256": "814bf2eca1d07018262a860819db8fc7ddcb7d1a866839caec53ef2993c625e6",
        "execution_scope": "Postprocessing selection only; full task setup not executed."
      },
      "primary_metric": "substring_exact_match",
      "official_judge_required": false,
      "schema": "mnemetric.memoryagentbench-answer-metrics/v1",
      "official_full_benchmark_run": false,
      "scope": "Answer scoring only; task-specific upstream postprocessing when sub_dataset is set. No agent execution or judge scoring. Summary and LongMemEval lexical metrics are auxiliary and do not replace their official LLM-judge evaluations.",
      "upstream_revision": "538026089d1a8a8eff05121d0db89b388f360eba",
      "source_sha256": "d77976be409298970614d477a9d8003850caddb0510e56a7e821a037d98493a2",
      "python": "3.12.14",
      "dependencies": {
        "numpy": "1.26.4",
        "nltk": "3.9.1",
        "tiktoken": "0.7.0",
        "rouge-score": "0.1.2",
        "editdistance": "0.8.1"
      },
      "cases": [
        {
          "case_id": "factconsolidation_sh_6k:0:0",
          "metrics": {
            "exact_match": false,
            "f1": 0,
            "substring_exact_match": false,
            "rougeL_f1": 0.0,
            "rougeL_recall": 0.0,
            "rougeLsum_f1": 0.0,
            "rougeLsum_recall": 0.0
          },
          "postprocessing": {
            "parsed_output": "goaltender is associated with the sport of ice"
          }
        }
      ]
    }
  },
  "limitations": [
    "Only one case, chosen by context screening before generation.",
    "Qwen substitutes for official GPT-4o-mini.",
    "BM25 response hit the unchanged 10-token cap; raw capped response scored without editing.",
    "No population-level quality or superiority conclusion."
  ]
}
