{
  "published": "2026-09-05",
  "cookEvaluationPeriod": "August 2026",
  "browserSource": "Cook Browser evaluation record, August 5, 2026",
  "memorySource": "Cook Memory research notes and evaluation records, August 2026",
  "externalSources": [
    "https://github.com/at-inc/aside-benchmarks",
    "https://github.com/rohitg00/agentmemory/blob/main/benchmark/LONGMEMEVAL.md"
  ],
  "browser": [
    {
      "id": "mind2web",
      "label": "Online-Mind2Web",
      "metric": "Task success \u00b7 %",
      "note": "Cook: 268/287 after 13 impossible tasks were excluded; 268/300 (89.3%) across the full cohort. Merged baseline and targeted reruns. Aside: 297/300. Different runs and scoring policies; published reference, not a controlled head-to-head. Other competitors reproduce the reference benchmark panel; evaluation setups differ.",
      "rows": [
        {
          "name": "Aside",
          "detail": "Published \u00b7 300 tasks",
          "score": 99,
          "display": "99.0%",
          "logo": "/logos/aside.svg"
        },
        {
          "name": "Browser Use",
          "score": 97.7,
          "display": "97.7%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-browserUseMark.svg"
        },
        {
          "name": "Cook Browser",
          "detail": "Desktop \u00b7 adjusted cohort",
          "score": 93.4,
          "display": "93.4%",
          "us": true,
          "logo": "/cook-mark.svg"
        },
        {
          "name": "GPT-5.4",
          "score": 92.8,
          "display": "92.8%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-openaiMark.svg"
        },
        {
          "name": "Claude Opus 4.8",
          "score": 84,
          "display": "84.0%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-claudeMark.svg"
        },
        {
          "name": "ChatGPT Atlas",
          "score": 70,
          "display": "70.0%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-atlasMark.svg"
        }
      ],
      "comparisonSource": "Archived reference benchmark results; Aside result from its published benchmark repository."
    },
    {
      "id": "bu",
      "label": "BU Bench V1",
      "metric": "Task success \u00b7 %",
      "note": "Cook: 17/18 after two invalid frozen answers were excluded; 17/20 (85.0%) on the frozen subset. Aside: 93/100 on the broader benchmark. Different sample sizes; these are published reference results, not a controlled ranking. Other competitors reproduce the reference benchmark panel; evaluation setups differ.",
      "rows": [
        {
          "name": "Cook Browser",
          "detail": "Desktop \u00b7 adjusted subset",
          "score": 94.4,
          "display": "94.4%",
          "us": true,
          "logo": "/cook-mark.svg"
        },
        {
          "name": "Aside",
          "detail": "Published \u00b7 100 tasks",
          "score": 93,
          "display": "93.0%",
          "logo": "/logos/aside.svg"
        },
        {
          "name": "Browser Use",
          "score": 89.5,
          "display": "89.5%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-browserUseMark.svg"
        },
        {
          "name": "Claude Fable 5",
          "score": 80,
          "display": "80.0%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-claudeMark.svg"
        },
        {
          "name": "GPT-5.5",
          "score": 68,
          "display": "68.0%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-openaiMark.svg"
        },
        {
          "name": "Gemini 3.6 Flash",
          "score": 58,
          "display": "58.0%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-geminiMark.svg"
        }
      ],
      "comparisonSource": "Archived reference benchmark results; Aside result from its published benchmark repository."
    },
    {
      "id": "odysseys",
      "label": "Odysseys",
      "metric": "Rubric score \u00b7 %",
      "note": "Cook: 34.5/40 rubric points after five unavailable tasks were excluded; strict score 39/47 (82.98%). Aside: 1,050/1,182 rubric items across 200 tasks. Cohorts and rubric aggregation differ; these are context, not equivalent evaluations. Other competitors reproduce the reference benchmark panel; evaluation setups differ.",
      "rows": [
        {
          "name": "Aside",
          "detail": "Published \u00b7 rubric items",
          "score": 88.8,
          "display": "88.8%",
          "logo": "/logos/aside.svg"
        },
        {
          "name": "Cook Browser",
          "detail": "Desktop \u00b7 adjusted points",
          "score": 86.25,
          "display": "86.25%",
          "us": true,
          "logo": "/cook-mark.svg"
        },
        {
          "name": "Browser Use",
          "score": 70,
          "display": "70.0%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-browserUseMark.svg"
        },
        {
          "name": "WebWright",
          "score": 60.8,
          "display": "60.8%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-webwrightMark.svg"
        },
        {
          "name": "Claude Opus 4.6",
          "score": 44.5,
          "display": "44.5%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-claudeMark.svg"
        },
        {
          "name": "GPT-5.4",
          "score": 33.5,
          "display": "33.5%",
          "detail": "Reference benchmark panel",
          "logo": "/logos/benchmark-openaiMark.svg"
        }
      ],
      "comparisonSource": "Archived reference benchmark results; Aside result from its published benchmark repository."
    }
  ],
  "memory": [
    {
      "id": "recall5",
      "label": "Recall @ 5",
      "metric": "Questions with a relevant result in the top 5 \u00b7 %",
      "note": "Cook: 476 answerable questions, scored on answer-bearing excerpts. agentmemory: published 500-question session-retrieval evaluation. Different retrieval units and cohorts; reference comparison, not an identical-harness rerun. Retrieval is separate from final-answer accuracy.",
      "rows": [
        {
          "name": "Cook Memory",
          "detail": "Answer-bearing excerpts",
          "score": 98.5,
          "display": "98.5%",
          "us": true
        },
        {
          "name": "agentmemory",
          "detail": "Published \u00b7 gold sessions",
          "score": 95.2,
          "display": "95.2%"
        }
      ]
    },
    {
      "id": "recall10",
      "label": "Recall @ 10",
      "metric": "Questions with a relevant result in the top 10 \u00b7 %",
      "note": "Cook: 476 answerable questions. agentmemory: published 500-question evaluation. Cook retrieves exchanges; agentmemory retrieves sessions. Scores measure finding relevant evidence, not whether the final answer is correct.",
      "rows": [
        {
          "name": "Cook Memory",
          "detail": "Answer-bearing excerpts",
          "score": 98.9,
          "display": "98.9%",
          "us": true
        },
        {
          "name": "agentmemory",
          "detail": "Published \u00b7 gold sessions",
          "score": 98.6,
          "display": "98.6%"
        }
      ]
    },
    {
      "id": "mrr",
      "label": "Rank quality",
      "metric": "Mean reciprocal rank \u00b7 0\u2013100, higher is better",
      "note": "MRR rewards placing the first relevant result earlier: rank 1 contributes 1, rank 2 contributes 1/2. Shown \u00d7100. Cook uses answer-bearing excerpts; agentmemory uses gold sessions. Published results from separate evaluation setups.",
      "rows": [
        {
          "name": "Cook Memory",
          "detail": "Answer-bearing excerpts",
          "score": 90.2,
          "display": "90.2",
          "us": true
        },
        {
          "name": "agentmemory",
          "detail": "Published \u00b7 gold sessions",
          "score": 88.2,
          "display": "88.2"
        }
      ]
    }
  ]
}
