{
  "success": true,
  "status": "completed",
  "outputReady": true,
  "output": {
    "text": "Catalogue of 54 qualifying benchmarks and evaluation frameworks, plus 30 adjacent, uncertain or excluded entries. Records distinguish release evidence, public artifacts, judging methods, reproducibility barriers and results provenance. No benchmark evaluations were run or reproduced.",
    "structured": {
      "scope": "Public benchmarks and evaluation frameworks with a documented release or substantive update from January 1, 2025 through September 26, 2026. Includes difficult web discovery, comprehensive entity/list retrieval, evidence-supported enrichment, literature discovery and research-report evaluation. Controlled web-derived corpora are included when they test research-agent workflows. A public paper or protocol can qualify even when tasks or scoring code are withheld; availability is assessed separately. Browser-action tasks, generic QA, preference-only search chat, internal-enterprise tasks and vendor run reports are separated. This is broad documented coverage, not a claim of exhaustiveness.",
      "summary": "The catalogue contains 54 qualifying benchmarks/frameworks and 30 adjacent, uncertain or excluded entries. Three distinct capabilities recur: finding a difficult answer, discovering and enriching an entire set, and producing a supported research report. Their scores are not interchangeable: [WideSearch](https://github.com/ByteDance-Seed/WideSearch) evaluates tables and set coverage, while [DeepResearch Bench](https://github.com/Ayanami0730/deep_research_bench) separates report quality from citation support. Public availability ranges from released tasks and evaluators to encrypted, partially public or paper-only protocols. [BrowseComp-Plus](https://github.com/texttron/BrowseComp-Plus) controls web-state variation with a fixed corpus; live-web benchmarks still depend on retrieval, provider and judge versions. Most inspected results are benchmark-author runs, often also vendor runs. Reka’s third-party extension is identified separately without assuming organizational independence. Renames, forks and run reports are not counted as new benchmarks. No benchmark code was executed and no reported result was reproduced.",
      "benchmarks": [
        {
          "name": "BrowseComp",
          "category": "difficult web discovery / short-answer research",
          "evidence": [
            {
              "url": "https://openai.com/index/browsecomp/",
              "supports": "Dated release, task construction, size, original model comparisons and leakage warning."
            },
            {
              "url": "https://github.com/openai/simple-evals/blob/main/browsecomp_eval.py",
              "supports": "Observed task-loading, decryption and model-grading implementation."
            },
            {
              "url": "https://arxiv.org/abs/2504.12516",
              "supports": "Primary paper linked from the official announcement."
            }
          ],
          "measures": "Tests persistent browsing and reasoning for obscure, entangled facts across 1,266 questions with short, verifiable answers; it is not an exhaustive-list or report-quality test.",
          "task_data": "Public test CSV is referenced by the evaluator; questions and answers are obfuscated using canary-based decoding.",
          "canonical_url": "https://openai.com/index/browsecomp/",
          "evaluation_code": "Public scripts in openai/simple-evals load and decode tasks, then invoke an LLM grader.",
          "judging_criteria": "An LLM judges reference-answer equivalence, producing aggregate correctness or accuracy rather than source-quality or list-completeness scores.",
          "reported_results_provenance": "Launch results are benchmark-author and vendor-reported by OpenAI, not independently reproduced.",
          "reproducibility_and_barriers": "Requires an agent/search implementation and grader access; live-web changes, model settings, and browsing effort affect results. Public answers create leakage risk, and authors request that decoded examples not be republished.",
          "eligibility_date_and_evidence": "2025-04-10: OpenAI publicly announced and open-sourced the benchmark."
        },
        {
          "name": "WideSearch: Benchmarking Agentic Broad Info-Seeking",
          "category": "comprehensive list enumeration + evidence-supported entity enrichment",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2508.07999",
              "supports": "Primary paper: benchmark purpose, 200 bilingual tasks, curation, metrics, tools, judging and reported runs."
            },
            {
              "url": "https://github.com/ByteDance-Seed/WideSearch",
              "supports": "Primary repository: 2025/08/11 release notice, evaluation code, instructions and dataset/project links."
            },
            {
              "url": "https://widesearch-seed.github.io/",
              "supports": "Official project page: full benchmark dataset/ground-truth tables download and evaluation-code access."
            },
            {
              "url": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch",
              "supports": "Official dataset artifact linked by the project/repository."
            }
          ],
          "measures": "Measures item, column, and row precision/recall; table/task success requires complete, accurate atomic information. Max@N and human comparison are also reported.",
          "task_data": "Public 200-task dataset: 100 English and 100 Chinese tasks across 18 domains. Agents fill predefined tables using large-scale atomic facts about multiple entities; curation included exhaustive human gold research and automated-versus-human validation.",
          "canonical_url": "https://arxiv.org/abs/2508.07999",
          "evaluation_code": "Public MIT repository contains evaluation scripts, documentation, and agent/tool code; the public Hugging Face dataset and project page link to data and code.",
          "judging_criteria": "Entity/item precision-recall/F1, attribute correctness and exact complete-row/table correctness; the automated evaluator was compared with expert ratings.",
          "reported_results_provenance": "Original authors report benchmark runs across 10+ agentic systems and human tests; results are author-reported, not independently validated.",
          "reproducibility_and_barriers": "Code and dataset are public, but live-web execution requires API credentials and results may drift with web, model, and API changes. Commercial-system tests used web interfaces; this is an assessment of barriers.",
          "eligibility_date_and_evidence": "Released August 11, 2025 (arXiv 2508.07999; paper version dated August 28/September 5 in rendered materials); the GitHub README announces release on 2025/08/11."
        },
        {
          "name": "Ko-WideSearch: A Korean Breadth-Search Benchmark for Exhaustive Set Enumeration by Web Agents",
          "category": "comprehensive list enumeration + evidence-supported entity enrichment",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2606.27595",
              "supports": "Primary paper: task construction, 228-table composition, tiers, cross-source enrichment, metrics, comparator, gating and author results."
            },
            {
              "url": "https://github.com/minstar/Ko-widesearch",
              "supports": "Official code repository: open pipeline/scorer, project citation and dataset/release information."
            },
            {
              "url": "https://minstar.github.io/Ko-widesearch/",
              "supports": "Official project page: benchmark scope, metrics, table counts, categories and results."
            },
            {
              "url": "https://huggingface.co/datasets/Minbyul/Ko-widesearch",
              "supports": "Official dataset card: 228-task schema, encrypted question/answer fields, tiers and scoring format."
            }
          ],
          "measures": "Measures membership Item-F1, attribute-cell Column-F1, complete-row Row-F1, table success, and parse rate using normalization-aware comparison.",
          "task_data": "Public schema/metadata with gated/encrypted answer fields: 228 Korean tables cover 190 parent entities and 16 categories, including 4,262 gold rows and 14,560 attribute cells. Easy/Medium/Hard tiers vary table width and composite-key membership; 201 tables require cross-source attribute lookups.",
          "canonical_url": "https://arxiv.org/abs/2606.27595",
          "evaluation_code": "Public MIT-licensed pipeline and scorer are available in the official repository; answer fields remain encrypted or gated by request to reduce leakage.",
          "judging_criteria": "Scores membership precision/recall, per-column cell correctness, strict full-row correctness, and whole-table success; the comparator normalizes name variants, date granularity, and numeric formatting.",
          "reported_results_provenance": "Original authors report a 20-agent evaluation; these are benchmark-author runs, not independently validated results.",
          "reproducibility_and_barriers": "Scorer and pipeline are public, but evaluation answers are gated/encrypted and live-web tasks are dynamic; API/model access and request-based data release remain practical barriers (assessment).",
          "eligibility_date_and_evidence": "2026-06: arXiv release 2606.27595; official project, repository and dataset are public."
        },
        {
          "name": "EnterList (introduced in WebLists)",
          "category": "Structured list extraction from interactive websites; paper-defined benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2504.12682",
              "supports": "Primary paper: date, 200 tasks/50 sites, reference-script design, live-ground-truth caveat, metrics and results."
            },
            {
              "url": "https://doi.org/10.48550/arxiv.2504.12682",
              "supports": "Primary DOI/arXiv record confirming 2025-04-17 publication and benchmark definition."
            }
          ],
          "measures": "Precision and recall for rows extracted from live interactive websites, with cost per output row; narrower than unconstrained open-web enumeration.",
          "task_data": "Availability not established as a public release: the paper defines 200 live tasks across four use cases and 50 websites, with annotated reference URLs and extraction scripts, capped at five pages per task.",
          "canonical_url": "https://arxiv.org/html/2504.12682",
          "evaluation_code": "The paper describes reference extraction scripts and methodology, but a public repository or downloadable evaluator/data package was not established; treat it as paper-only.",
          "judging_criteria": "Exact field matching against refreshed reference extraction defines precision as retrieved gold rows and recall as retrieved gold coverage; live-site ground truth changes over time.",
          "reported_results_provenance": "Original authors report experiments including BardeenAgent results; no independent reproduction was established.",
          "reproducibility_and_barriers": "Live websites and changing ground truth hinder reproduction; no verified public task, data, or evaluator release was established.",
          "eligibility_date_and_evidence": "ArXiv record dated 2025-04-17, within the eligibility window."
        },
        {
          "name": "DeepResearch Bench (Ayanami0730 / Mingxuan Du; RACE + FACT)",
          "category": "long-form deep-research report benchmark; citation/retrieval evaluation",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2506.11763",
              "supports": "Paper identity, 100 tasks, RACE/FACT methodology, release link and submission date."
            },
            {
              "url": "https://github.com/Ayanami0730/deep_research_bench",
              "supports": "Official repository, code/data/result layout, update history and evaluation instructions."
            },
            {
              "url": "https://deepresearch-bench.github.io/",
              "supports": "Official project description, task composition and author-reported leaderboard/results."
            }
          ],
          "measures": "RACE measures report quality through comprehensiveness, depth, instruction following, and readability; FACT measures citation accuracy and effective citation count.",
          "task_data": "Public repository materials include 100 PhD-level tasks, 50 Chinese and 50 English, across 22 fields, authored or refined by 100+ domain experts. It includes task, reference, report, and result materials, with raw articles and scores linked on the leaderboard.",
          "canonical_url": "https://github.com/Ayanami0730/deep_research_bench",
          "evaluation_code": "Public repository contains RACE and FACT evaluation code, prompts, configurations, and result directories; execution requires model/API and web-content retrieval services.",
          "judging_criteria": "RACE uses LLM-generated task criteria and weights to compare reports with high-quality references, with human-consistency validation. FACT extracts statement-URL pairs, retrieves cited pages, and LLM-judges support.",
          "reported_results_provenance": "Authors report evaluations of commercial deep-research agents and search-enabled LLMs; linked raw articles and scores remain author-run, not independent results.",
          "reproducibility_and_barriers": "Code and data are publicly downloadable under Apache-2.0, but reproducing scores requires paid or credentialed judge models and web retrieval; live URLs may change. These are assessed barriers.",
          "eligibility_date_and_evidence": "2025-06: paper arXiv:2506.11763 and public repository (created June 13) introduce RACE/FACT evaluation. Distinct from FutureSearch’s similarly named benchmark."
        },
        {
          "name": "DeepResearch Bench II (imlrz / Ruizhe Li et al.)",
          "category": "long-form deep-research report benchmark; expert-rubric diagnosis",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2601.08536",
              "supports": "Paper identity/date, 132 tasks, 9,430 rubrics, dimensions, human review and release claims."
            },
            {
              "url": "https://github.com/imlrz/DeepResearch-Bench-II",
              "supports": "Official repo, November 2025 release, Apache-2.0 status and evaluator artifacts."
            },
            {
              "url": "https://arxiv.org/html/2601.08536",
              "supports": "Benchmark abstract and rubric construction/evaluation description."
            }
          ],
          "measures": "Scores binary satisfaction of information-recall, analysis and presentation criteria across 9,430 fine-grained rubrics and 132 tasks.",
          "task_data": "Public tasks_and_rubrics.jsonl contains 132 tasks and 9,430 criteria across 22 domains. A February 24, 2026 Hugging Face release adds selected model-generated reports. May 2026 metadata assigns each task its source article’s license.",
          "canonical_url": "https://github.com/imlrz/DeepResearch-Bench-II",
          "evaluation_code": "Official Apache-2.0 repository provides rubrics, scripts and a batched evaluator for PDF, DOCX, image and text reports. Running requires Gemini or other model credentials.",
          "judging_criteria": "LLM judges assess binary satisfaction of atomic information-recall, analysis and presentation criteria. Criteria were extracted from expert reports and human-reviewed.",
          "reported_results_provenance": "The paper reports author evaluations of several state-of-the-art agents; no independent result provenance is established here.",
          "reproducibility_and_barriers": "Benchmark, rubrics and code are public; assessment: judge-model access, processing cost, evaluator versions and report formatting affect reproducibility.",
          "eligibility_date_and_evidence": "Official README records a November 2025 pipeline release; arXiv:2601.08536 was submitted 2026-01-13. It is explicitly a follow-up to DeepResearch Bench."
        },
        {
          "name": "DeepResearch-ReportEval (HKUDS/DeepResearch-Eval)",
          "category": "long-form research-report evaluation framework/dataset",
          "evidence": [
            {
              "url": "https://github.com/HKUDS/DeepResearch-Eval",
              "supports": "Official scripts, data structure, dimensions, fact labels and dataset description."
            },
            {
              "url": "https://doi.org/10.48550/arxiv.2510.07861",
              "supports": "Paper date, framework identity, 100-query dataset, methodology and author-reported results."
            },
            {
              "url": "https://arxiv.org/pdf/2510.07861",
              "supports": "Primary source for quality, redundancy, factuality and expert-alignment claims."
            }
          ],
          "measures": "Scores comprehensiveness, coherence, clarity, insightfulness and overall quality, plus paragraph redundancy and citation-supported factuality.",
          "task_data": "100 queries cover 12 real-world categories and include 100 Qwen-DeepResearch reports collected in early September 2025; repository data and examples are public.",
          "canonical_url": "https://github.com/HKUDS/DeepResearch-Eval",
          "evaluation_code": "Repository includes judge_score.py, judge_fact.py, utilities, prompts and examples. Fact checking uses Firecrawl or Jina Reader with -1/0/1 support labels.",
          "judging_criteria": "LLM judges score report quality and paragraph redundancy; a citation checker evaluates each claim as unsupported, uncertain or supported using retrieved source pages.",
          "reported_results_provenance": "Paper comparisons cover four commercial systems and author-generated Qwen reports; no independent reproduction is identified here.",
          "reproducibility_and_barriers": "MIT code and data are public; assessment: judge-model and Firecrawl/Jina credentials, web access and model selection are practical barriers. The dataset is a fixed snapshot.",
          "eligibility_date_and_evidence": "Paper arXiv:2510.07861 was published 2025-10-09, and the official repository identifies the framework and dataset; it qualifies within the window."
        },
        {
          "name": "ResearcherBench",
          "category": "Scientific web research and evidence-grounded report generation",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2507.16280",
              "supports": "Paper identity/date, 65 questions, 35 subjects, metrics and release."
            },
            {
              "url": "https://github.com/GAIR-NLP/ResearcherBench",
              "supports": "Official code, data format, evaluation workflow and API requirements."
            },
            {
              "url": "https://researcherbench.github.io/",
              "supports": "Task construction, methodology, leaderboard provenance and evaluation timing."
            }
          ],
          "measures": "Scores weighted expert-insight coverage, citation-support accuracy (Faithfulness) and citation coverage (Groundedness).",
          "task_data": "Public repository provides 65 expert-curated questions across 35 AI subjects, with weighted reference insights; tasks include technical questions, literature reviews and research consulting.",
          "canonical_url": "https://github.com/GAIR-NLP/ResearcherBench",
          "evaluation_code": "Official repository provides the benchmark platform, formats and eval.sh; users submit model responses and receive rubric and factuality results. OpenAI and Jina credentials are required.",
          "judging_criteria": "Researchers create weighted 1–3 insight criteria, with Claude-3.7-Sonnet assisting extraction; a factuality pipeline extracts claims and URLs, then judges source support.",
          "reported_results_provenance": "Official paper and site report author evaluations of OpenAI, Gemini, Grok, Perplexity and other baselines; no independent runs are established.",
          "reproducibility_and_barriers": "Questions and framework are public; assessment: API credentials, web retrieval and expert rubric interpretation are barriers. Baselines are author-run and dated March–June 2025.",
          "eligibility_date_and_evidence": "Paper arXiv:2507.16280 was published 2025-07-22; the official site and repository are public within the window. It evaluates frontier-AI scientific questions rather than broad PhD-level web research."
        },
        {
          "name": "ResearchRubrics",
          "category": "Deep-research report evaluation / human-authored rubrics",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2511.07685",
              "supports": "Benchmark size, human-authored rubrics, dimensions, systems and reported compliance."
            },
            {
              "url": "https://github.com/scaleapi/researchrubrics",
              "supports": "Public MIT code, data expectations and executable evaluation pipeline."
            }
          ],
          "measures": "Measures weighted criterion compliance across 101 prompts and 2,593 expert-written criteria, plus agreement between human and model judgments.",
          "task_data": "Public ScaleAI/researchrubrics download is documented in the repository: 101 human-written prompts and 2,593 reviewed criteria across nine domains, including penalty criteria.",
          "canonical_url": "https://github.com/scaleapi/researchrubrics",
          "evaluation_code": "Actual MIT-licensed code ingests Markdown reports, chunks them, performs batch evaluation and calculates compliance. Default Gemini 2.5 Pro use through LiteLLM requires an API key and ScaleAI Hugging Face data.",
          "judging_criteria": "Human criteria cover requirements, reasoning, synthesis, references and communication. The released evaluator assigns Satisfied/Not Satisfied scores and computes weighted compliance; negative-weight rubrics are excluded from the denominator.",
          "reported_results_provenance": "Scale AI authors evaluated OpenAI, Gemini and Perplexity Deep Research; no independent leaderboard or result is found in the checked primary artifacts.",
          "reproducibility_and_barriers": "Prompts, rubrics and code are public; external model access, cost and latency remain barriers. Commercial reports from the study may not be reproducible from the repository.",
          "eligibility_date_and_evidence": "Paper/preprint arXiv:2511.07685 was published 2025-11-10 and the repository was created 2025-11-08; both fall within the window. Repository and Hugging Face data are public."
        },
        {
          "name": "LiveResearchBench + DeepEval",
          "category": "Dynamic web research benchmark and long-form report evaluator",
          "evidence": [
            {
              "url": "https://github.com/SalesforceAIResearch/LiveResearchBench",
              "supports": "Public benchmark and DeepEval code, protocols, tasks and outputs."
            },
            {
              "url": "https://arxiv.org/abs/2510.14240",
              "supports": "Release date, design, 100 tasks, research effort and 17-system evaluation."
            },
            {
              "url": "https://huggingface.co/datasets/Salesforce/LiveResearchBench",
              "supports": "Public static/realtime datasets and date-placeholder behavior."
            },
            {
              "url": "https://livedeepresearch.github.io/",
              "supports": "Public leaderboard and reported system-level scores."
            }
          ],
          "measures": "Scores coverage, presentation, citation accuracy, citation traceability, fact-and-logic consistency and analysis depth using checklist, pointwise, pairwise and ensemble protocols.",
          "task_data": "Public static and realtime Hugging Face datasets: 100 expert-curated tasks across seven domains and ten categories, with checklists and dynamic date placeholders.",
          "canonical_url": "https://github.com/SalesforceAIResearch/LiveResearchBench",
          "evaluation_code": "Actual Salesforce code provides benchmark loading, DeepEval protocols, report handling and result output; the Hugging Face dataset is public. Enterprise-Deep-Research documents generation and invocation.",
          "judging_criteria": "Human checklists score coverage and presentation; rubric trees assess citation correctness and association; consistency and pairwise protocols assess factual logic and comparative depth.",
          "reported_results_provenance": "Salesforce-led evaluation covers 17 frontier systems; public leaderboard values are author/vendor results, not independent evaluations.",
          "reproducibility_and_barriers": "Tasks, code and static data are public; assessment: realtime evaluation depends on live web content, while generation and LLM judging require provider APIs.",
          "eligibility_date_and_evidence": "Paper arXiv:2510.14240 was published in October 2025; the GitHub repository is dated 2025-10-17, and an ICLR 2026 artifact is public."
        },
        {
          "name": "ReportBench",
          "category": "Academic survey-grounded report quality and citation evaluation",
          "evidence": [
            {
              "url": "https://github.com/ByteDance-BandAI/ReportBench",
              "supports": "Actual public code/data structure, evaluation workflow and full reported result table."
            },
            {
              "url": "https://openreview.net/forum?id=zvL42fmtbG",
              "supports": "Public conference record and abstract confirming benchmark purpose and public-release status."
            }
          ],
          "measures": "Measures citation precision, recall, citation-match rate, reference counts, cited-statement coverage, and non-cited factual accuracy using statement-level verification.",
          "task_data": "Public ReportBench_v1.1.jsonl supplies 100 survey-grounded research tasks across ten domains and three prompt granularities; survey references furnish citation gold.",
          "canonical_url": "https://github.com/ByteDance-BandAI/ReportBench",
          "evaluation_code": "Public repository contains processing, retrieval, citation and statement evaluators, metric calculators, benchmark data directories, and JSON-output support. Commercial web-product collection requires browser captures and dedicated processors.",
          "judging_criteria": "Published survey references are gold standards; cited claims are matched to retrieved passages, while non-cited claims use web-connected Gemini majority voting.",
          "reported_results_provenance": "ByteDance BandAI author-run results report precision, recall, citation matching, and non-cited accuracy for OpenAI and Gemini Deep Research. No independent result was located in checked primary artifacts.",
          "reproducibility_and_barriers": "Code and data are public, but web retrieval, browser capture, paid LLMs, and paid search services are required. Survey overlap may penalize valid divergent research; this is a methodological limitation assessment.",
          "eligibility_date_and_evidence": "arXiv:2508.15804 posted August 2025; GitHub repository public by the 2025 paper release, within window."
        },
        {
          "name": "FACTS Search (FACTS Benchmark Suite)",
          "category": "Adjacent web-search factuality benchmark",
          "evidence": [
            {
              "url": "https://deepmind.google/blog/facts-benchmark-suite-systematically-evaluating-the-factuality-of-large-language-models/",
              "supports": "Official December 2025 release, Search purpose, public/private sizes and suite leaderboard."
            },
            {
              "url": "https://arxiv.org/abs/2512.10791",
              "supports": "Search dataset composition, Brave API, metrics and reported results."
            },
            {
              "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard",
              "supports": "Public live Search leaderboard and operational evaluation description."
            }
          ],
          "measures": "Measures search-enabled answer F1, overall and attempted accuracy, hedging rate, and search count. A prompted auto-rater scores answer correctness.",
          "task_data": "1,884 Search questions comprise 890 public and 994 private items, including human-written hard-tail questions and three synthetic multi-hop subsets. The task evaluates factual search answers, not reports or list-building.",
          "canonical_url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard",
          "evaluation_code": "Public benchmark examples and hosted Kaggle evaluation are available, but exact leaderboard reproduction requires the standardized Brave API, private set, hosted judge, and LLM setup.",
          "judging_criteria": "A prompted auto-rater compares answers with gold answers; F1 balances accuracy and attempted accuracy, while hedging and search count are diagnostics.",
          "reported_results_provenance": "Google DeepMind, Google Research, and Kaggle author-hosted evaluation reports results for Gemini 3 Pro, GPT-5, and Claude 4.5 Opus. No independent result was counted.",
          "reproducibility_and_barriers": "Public examples and a common search API improve comparability, but the private set, API costs, availability, and auto-rater introduce barriers and possible bias. This is adjacent because it evaluates factual search answers.",
          "eligibility_date_and_evidence": "FACTS Benchmark Suite and Search benchmark launched December 2025 through the official announcement and arXiv:2512.10791; leaderboard remained public through 2026."
        },
        {
          "name": "BrowseComp-Plus",
          "category": "fixed-corpus deep-research benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2508.06600v1",
              "supports": "Paper date, fixed curated corpus, 830 queries, metrics, and reported results."
            },
            {
              "url": "https://github.com/texttron/BrowseComp-Plus",
              "supports": "Official code, reproduction scripts, qrels/evaluation instructions, and static-corpus description."
            },
            {
              "url": "https://texttron.github.io/BrowseComp-Plus/",
              "supports": "Official project description and metric definitions."
            }
          ],
          "measures": "Measures answer accuracy, evidence-document recall, search-call count, calibration error, retrieval Recall@k, and nDCG@10.",
          "task_data": "Public fixed corpus, queries and qrels: 830 BrowseComp-derived questions over approximately 100,000 web documents, with human-verified evidence and hard negatives. This is a controlled web-derived corpus, not live search.",
          "canonical_url": "https://github.com/texttron/BrowseComp-Plus",
          "evaluation_code": "Public official repository provides agent and retriever scripts, evaluation scripts, TREC-style qrels, reproduction documentation, and leaderboard instructions. Main evaluation uses top-five retrieval with a 512-token context limit.",
          "judging_criteria": "GPT-4.1 judges final-answer correctness; labeled evidence and gold documents determine retrieval Recall and nDCG. Citation accuracy is analyzed separately.",
          "reported_results_provenance": "Paper and official project results are author-reported comparisons of retrieval and model systems, including Search-R1, GPT-5, and GPT-5 with Qwen3 embeddings; no independent results are reported.",
          "reproducibility_and_barriers": "The static corpus and released qrels improve reproducibility, while model/API access and substantial compute remain barriers. Dataset and retrieval artifacts are public through repository and Hugging Face links.",
          "eligibility_date_and_evidence": "Public paper and repository released 2025-08-08 (arXiv:2508.06600; GitHub created/updated in 2025), within window."
        },
        {
          "name": "BrowseComp-ZH",
          "category": "Chinese live-web multi-hop browsing benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2504.19314",
              "supports": "Paper date, 289-question design, domains, construction, live-web task and reported evaluations."
            },
            {
              "url": "https://github.com/PALIN2018/BrowseComp-ZH",
              "supports": "Official dataset/code, encrypted-data procedure, validation and evaluation README."
            },
            {
              "url": "https://github.com/open-compass/AgentCompass/blob/a7c30989/src/agentcompass/benchmarks/browsecomp_zh.py",
              "supports": "Public evaluator integration showing dataset loading and LLM judge scoring."
            },
            {
              "url": "https://github.com/AGI-Eval-Official/BrowseComp-ZH-revised",
              "supports": "Third-party correction fork, January 2026 creation, 24 corrections and decoding instructions; not official benchmark-author validation."
            }
          ],
          "measures": "Measures answer accuracy and calibration error for standalone LLMs and browsing agents, including model, reasoning, and browsing comparisons.",
          "task_data": "Public encrypted dataset of 289 Chinese multi-hop questions across 11 domains; repository decoding is required. The January 2026 AGI-Eval correction fork claims 24 answer corrections, so versions are not interchangeable.",
          "canonical_url": "https://github.com/PALIN2018/BrowseComp-ZH",
          "evaluation_code": "Public official repository includes decryption, model-evaluation, prediction/result, and calibration workflows. OpenCompass integration loads the dataset and uses an LLM judge.",
          "judging_criteria": "An LLM scorer judges answer correctness; the OpenCompass implementation explicitly uses LLMJudgeScorer. Official results report accuracy and calibration error.",
          "reported_results_provenance": "Official author-run comparisons cover more than 20 systems, including OpenAI DeepResearch, O1, and Gemini-2.5-Pro. The reported figures are benchmark-author results.",
          "reproducibility_and_barriers": "Encrypted questions and answers require the repository decryption procedure; live Chinese web access and proprietary model APIs are practical barriers. Encryption limits direct dataset inspection.",
          "eligibility_date_and_evidence": "2025-04-24: original BrowseComp-ZH paper and repository. A separate AGI-Eval correction fork appeared in January 2026."
        },
        {
          "name": "WebWalkerQA",
          "category": "web traversal / information-seeking QA benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2501.07572",
              "supports": "Official paper, benchmark size/task formulation, traversal setup, GPT-4 judging and metrics."
            },
            {
              "url": "https://aclanthology.org/2025.acl-long.508.pdf",
              "supports": "ACL paper details, dataset construction, 680 QA pairs, 1,373 pages, bilingual/domain statistics and evaluation."
            },
            {
              "url": "https://github.com/Alibaba-NLP/WebAgent",
              "supports": "Official implementation/project and benchmark release links."
            },
            {
              "url": "https://huggingface.co/datasets/callanwu/WebWalkerQA",
              "supports": "Public dataset card describing 680 QA records and fields including root URL, hop and golden path."
            }
          ],
          "measures": "Measures question-answer accuracy and action count, with analyses by traversal depth, source count, domain, and language. Action count measures efficiency.",
          "task_data": "Public Hugging Face dataset: 680 Chinese/English questions over 1,373 webpages, with root URLs, hop counts and golden paths; tasks require single-site traversal or multisource navigation.",
          "canonical_url": "https://github.com/Alibaba-NLP/WebAgent",
          "evaluation_code": "Public official WebAgent repository provides the WebWalker framework, demo, and benchmark materials; the dataset is publicly distributed through Hugging Face. The paper specifies click-only interaction and a 15-step limit.",
          "judging_criteria": "GPT-4 evaluates answer correctness because generated answers vary, making exact match unsuitable. Correct-run action count measures efficiency.",
          "reported_results_provenance": "Original WebWalker results are benchmark-author runs. WebDancer reports separate WebWalkerQA runs; these are evaluations of the existing benchmark, not new benchmarks.",
          "reproducibility_and_barriers": "The public dataset and implementation support reproduction, but crawling, rendering, live-site changes, and web traversal create barriers. The benchmark requires access to current webpages.",
          "eligibility_date_and_evidence": "arXiv preprint 2501.07572 published 2025-01-13; official project announcement says WebWalker released 2025-01-14 and accepted ACL 2025."
        },
        {
          "name": "xbench-DeepSearch",
          "category": "live-web deep-search/tool-use benchmark",
          "evidence": [
            {
              "url": "https://github.com/xbench-ai/xbench-evals",
              "supports": "Official datasets, runner, releases, metrics and leaderboard tables."
            },
            {
              "url": "https://xbench.org/agi/aisearch",
              "supports": "Official scope, open-source links, update information and Eval Card link."
            },
            {
              "url": "https://xbench.org/files/Eval%20Card%20xbench-DeepSearch.pdf",
              "supports": "Evaluation design, validation, manual execution and LLM judging."
            },
            {
              "url": "https://github.com/xbench-ai/xbench-evals/blob/main/data/DeepSearch-2510.csv",
              "supports": "Released dataset schema and encrypted task artifact."
            }
          ],
          "measures": "Measures task accuracy and supports cost/task and time/task reporting; the public runner enables repeated evaluation.",
          "task_data": "Public repository provides encrypted CSV datasets for DeepSearch-2505 and DeepSearch-2510, covering planning, search, reasoning and summarization; updates are quarterly.",
          "canonical_url": "https://github.com/xbench-ai/xbench-evals",
          "evaluation_code": "Public repository includes xbench_evals.py, data artifacts and decryption/evaluation workflows; versions 2505/2510 and an LLM-judge scorer are inspectable.",
          "judging_criteria": "An LLM judge scores answers; questions underwent manual collection, human validation and quality filtering.",
          "reported_results_provenance": "Official leaderboard reports author-run provider/product evaluations, including May and August 2025 tables; no independent reproductions are established.",
          "reproducibility_and_barriers": "Encrypted files and proprietary agent interfaces limit full reproduction; public runner and data support model-side reproduction where APIs exist. Assessment: many product runs were manual.",
          "eligibility_date_and_evidence": "Official README published 2025-05-28; DeepSearch-2505 and DeepSearch-2510 releases fall within the eligibility window."
        },
        {
          "name": "SealQA",
          "category": "search-augmented factual reasoning benchmark",
          "evidence": [
            {
              "url": "https://doi.org/10.48550/arxiv.2506.01062",
              "supports": "Publication date, benchmark variants, task design, versioning and reported results."
            },
            {
              "url": "https://huggingface.co/datasets/vtllms/sealqa",
              "supports": "Official public dataset, splits, results and contamination canary."
            }
          ],
          "measures": "Measures factual-answer accuracy with and without search across Seal-0 and Seal-Hard; LongSeal measures evidence selection among distractor documents.",
          "task_data": "Public data include Seal-0, Seal-Hard and LongSeal, covering conflicting, noisy or unhelpful search results across domains; LongSeal supplies many documents with one relevant answer source.",
          "canonical_url": "https://huggingface.co/datasets/vtllms/sealqa",
          "evaluation_code": "Public Hugging Face data and paper artifacts are available; the benchmark is dynamic and periodically updated. No standalone official evaluator repository was verified.",
          "judging_criteria": "Scores factual answer accuracy under no-search and search/tool conditions. Exact automated judge implementation is not established from inspected sources.",
          "reported_results_provenance": "Reported results are author-run evaluations using specified models and search systems; no independent reproductions are established.",
          "reproducibility_and_barriers": "Public data aid access, but version updates, search-provider behavior and proprietary APIs limit frozen-test reproduction. A canary string addresses contamination.",
          "eligibility_date_and_evidence": "arXiv:2506.01062 was published 2025-06-01; the official dataset is publicly released within the eligibility window."
        },
        {
          "name": "Mind2Web 2",
          "category": "Core research benchmark: agentic search / long-horizon web research and citation-backed synthesis",
          "evidence": [
            {
              "url": "https://github.com/OSU-NLP-Group/Mind2Web-2",
              "supports": "Release dates, evaluation-script updates, repository and instructions."
            },
            {
              "url": "https://github.com/OSU-NLP-Group/Mind2Web-2/blob/main/run_eval.py",
              "supports": "Evaluation runner, version options and local execution."
            },
            {
              "url": "https://huggingface.co/datasets/osunlp/Mind2Web-2",
              "supports": "Dataset artifact and all-scripts release corroboration."
            },
            {
              "url": "https://arxiv.org/html/2506.21506v2",
              "supports": "Task design, metrics, rubric workflow and reported evaluations."
            }
          ],
          "measures": "Measures task completion through Partial Completion, Success Rate and Pass@3, alongside completion time, answer length, correctness and source attribution.",
          "task_data": "Public artifact provides 130 long-horizon web-search tasks involving synthesis, list retrieval, time-varying information and citations; dev and test data are available.",
          "canonical_url": "https://osu-nlp-group.github.io/Mind2Web-2/",
          "evaluation_code": "Public MIT repository includes run_eval.py and task-specific Extractor/Verifier workflows; the 2025-10-23 release covers both public dev and test sets.",
          "judging_criteria": "An Extractor parses claims and citations, while a Verifier checks them against webpages or screenshots; leaf scores aggregate into task and root metrics.",
          "reported_results_provenance": "Original authors report 2025 agent and human evaluations, including private-test-era results; no independent reproduction is established.",
          "reproducibility_and_barriers": "Public code and tasks improve reproduction, but live webpages, crawling, LLM/API judges and changing information introduce cost and nondeterminism. Assessment: results may drift.",
          "eligibility_date_and_evidence": "Initial public release was 2025-06-26; the 2025-10-23 update released evaluation scripts for both public dev and test sets, within the eligibility window."
        },
        {
          "name": "Deep Research Bench (DRB; FutureSearch)",
          "category": "Web research, dataset/list compilation and evidence discovery; frozen/live variants",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2506.06287v1",
              "supports": "Paper date, task scope, RetroSearch, withheld data and methodology."
            },
            {
              "url": "https://futuresearch.ai/deep-research-bench/",
              "supports": "Official benchmark page, later snapshot, frozen pages and update policy."
            },
            {
              "url": "https://drb.futuresearch.ai/",
              "supports": "Official public leaderboard and evaluation endpoint."
            }
          ],
          "measures": "Eight research task families include dataset compilation, reference-class enumeration, evidence gathering, source tracing, numeric derivation and claim validation. The paper has 89 task instances; later leaderboard snapshots differ.",
          "task_data": "Full tasks are explicitly withheld to limit contamination; the paper publishes eight examples. Human-worked references and a frozen RetroSearch corpus support controlled evaluation, but a complete public bundle was not verified.",
          "canonical_url": "https://arxiv.org/abs/2506.06287",
          "evaluation_code": "The paper describes agent tooling and automated trace evaluation, and a public leaderboard exists. No complete public task/evaluator repository was verified; full reproduction code availability is partial/unclear.",
          "judging_criteria": "Task-specific scoring: row precision/recall/F1; URL or evidence recall; binary numeric/source correctness; and normalized distance from human probability judgments. LLMs assist entity matching and source-reliability checks; trace diagnostics are separate.",
          "reported_results_provenance": "FutureSearch authors report evaluations of LLMs, agents and commercial research products; leaderboard results are vendor/author-run, not independent validation.",
          "reproducibility_and_barriers": "RetroSearch controls web drift, but withheld tasks, proprietary product runs and incomplete evaluator artifacts limit independent reproduction. Assessment: repeatability is partial.",
          "eligibility_date_and_evidence": "2025-06: paper arXiv:2506.06287 introduces the benchmark; the official FutureSearch post is dated June 25, 2025."
        },
        {
          "name": "DeepSearchQA (Google DeepMind)",
          "category": "comprehensive multi-entity/list retrieval and deep-search benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2601.20975v1",
              "supports": "Report scope, task construction, metrics, judging and results."
            },
            {
              "url": "https://huggingface.co/datasets/google/deepsearchqa",
              "supports": "Public dataset artifact, schema, answer types and evaluator guidance."
            },
            {
              "url": "https://www.kaggle.com/benchmarks/google/dsqa/leaderboard",
              "supports": "Official Google/Kaggle leaderboard endpoint."
            }
          ],
          "measures": "Measures exhaustive set-answer generation through fully correct and incorrect set rates, extraneous-answer rate and F1, covering collation, deduplication, entity resolution and stopping.",
          "task_data": "Public dataset contains 900 expert-annotated prompts across 17 fields with objectively verifiable gold sets, answer types and time or source anchors; about 65% are set-answer tasks.",
          "canonical_url": "https://arxiv.org/abs/2601.20975",
          "evaluation_code": "Public dataset is downloadable, but no official standalone evaluator repository was verified. Third-party runners exist and are not Google code.",
          "judging_criteria": "A fully correct response exactly matches the gold set; F1 balances missing and extraneous entities. Evaluation uses an automated judging methodology.",
          "reported_results_provenance": "Google DeepMind results are author-run; Kaggle results are platform-evaluated submissions and do not establish independent scientific replication.",
          "reproducibility_and_barriers": "Static or time-anchored tasks and public data aid reproduction; web access, APIs and judge-model choice remain barriers. Official evaluator-code availability is unknown.",
          "eligibility_date_and_evidence": "Technical report posted 2026-01-28, within the cutoff; the public dataset is available on Hugging Face and the official leaderboard is hosted by Kaggle."
        },
        {
          "name": "DRACO Benchmark (Perplexity Research)",
          "category": "long-form research report quality/citation benchmark",
          "evidence": [
            {
              "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark",
              "supports": "Official release: tasks, construction, rubrics, judge prompt, dimensions, results and limitations."
            },
            {
              "url": "https://arxiv.org/html/2602.11685v1",
              "supports": "Primary technical report dated 2026-02-12 and detailed DRACO methodology/results."
            },
            {
              "url": "https://hf.co/datasets/perplexity-ai/draco",
              "supports": "Public dataset location linked by the official release."
            }
          ],
          "measures": "Evaluates factual accuracy, breadth/depth, presentation quality, objectivity and citation quality across 100 tasks using weighted binary rubric criteria, including penalties for unsupported claims.",
          "task_data": "Public dataset availability: production-derived Perplexity Deep Research requests were de-identified, reformulated and filtered through five stages, with expert-reviewed rubrics.",
          "canonical_url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark",
          "evaluation_code": "Public benchmark, rubrics, judge prompt and linked dataset are released; the exact harness boundary is unclear, and proprietary production sampling and research tools remain unreproducible.",
          "judging_criteria": "An LLM judge assigns weighted binary verdicts to rubric criteria; rubrics assess accuracy, completeness/depth, presentation and primary-source citation quality, with reliability checked across three judge models.",
          "reported_results_provenance": "Perplexity's own comparison of four deep-research systems; reported scores and leadership claims are vendor-run, not independent.",
          "reproducibility_and_barriers": "Public tasks, rubrics and judge prompt support reproduction, but production-query sampling, proprietary tools and English single-turn scope remain barriers or limitations (assessment).",
          "eligibility_date_and_evidence": "Eligible: public release announced in 2026; technical report arXiv:2602.11685 is dated 2026-02-12, before the 2026-09-26 cutoff. The announcement says the benchmark, rubrics and judge prompt are open sourced."
        },
        {
          "name": "WANDR: A Benchmark for Wide and Deep Research",
          "category": "comprehensive multi-entity discovery plus evidence-supported enrichment (wide-and-deep research)",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2608.14747",
              "supports": "Primary paper: release, tasks, hierarchy, construction, scoring pipeline, metrics and system results."
            },
            {
              "url": "https://github.com/perplexityai/wandr",
              "supports": "Observed repository: tasks, Harbor packages, evaluator, configs, instructions and paid-API requirements."
            },
            {
              "url": "https://research.perplexity.ai/articles/wandr-benchmark-evaluating-research-agents-that-must-search-wide-and-deep",
              "supports": "Official description of evidence, dynamic judging, rollups and benchmark setup."
            }
          ],
          "measures": "Scores record-level and hierarchical precision, recall and F1, including hard complete-subtree scores, soft partial-credit scores, retrieval-only versus full-record results, and task rollups.",
          "task_data": "Public 500-task packages encode qualification hierarchies and requested record volumes. Packages include instructions, schemas, fixtures, labels and evaluator artifacts, but not a static exhaustive gold universe.",
          "canonical_url": "https://arxiv.org/abs/2608.14747",
          "evaluation_code": "Public repository contains source tasks, Harbor adapter, generated packages, task-local evaluator, scripts, configurations and reports, with validation and full-run instructions.",
          "judging_criteria": "The evaluator fetches and canonicalizes cited pages, deduplicates entities, verifies claims against pages and excerpts, then aggregates task-specific record verdicts hierarchically. Precision measures submitted quality; recall measures coverage against requested volume.",
          "reported_results_provenance": "Original authors' pinned runs over six production systems; these are author-reported benchmark results, not independent reproductions.",
          "reproducibility_and_barriers": "Public tasks, Docker packages, evaluator, configs and manifests support reproduction. End-to-end runs require live retrieval and paid APIs; web drift, bot walls, provider settings and judge nondeterminism remain barriers (assessment).",
          "eligibility_date_and_evidence": "Submitted to arXiv 2026-08-14, within the 2026-09-26 cutoff. The official repository README was available by 2026-07-14 and contains the released task/evaluator tree."
        },
        {
          "name": "DeepWide (Table-as-Search business-development benchmark)",
          "category": "constrained entity discovery and attribute enrichment; paper-defined evaluation",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2602.06724",
              "supports": "Primary paper: task definitions, 20-query data, references, judging, metrics, limitations and results."
            },
            {
              "url": "https://github.com/AIDC-AI/Marco-Search-Agent",
              "supports": "Observed repository: DeepWideSearch artifacts and Table-as-Search implementation, prompts and tools."
            },
            {
              "url": "https://arxiv.org/abs/2510.20168",
              "supports": "Distinguishes earlier 220-question DeepWideSearch from the 20-query evaluation."
            }
          ],
          "measures": "Column-F1 measures identification of entities satisfying complex constraints; Item-Precision measures correctness of retrieved attributes. Tasks use fixed retrieval quantities rather than an exhaustive-universe claim.",
          "task_data": "20 real-world business-development and e-commerce queries require constrained discovery and attribute enrichment. Expert-verified references combine pooled system matches; a complete downloadable bundle was not separately verified.",
          "canonical_url": "https://arxiv.org/abs/2602.06724",
          "evaluation_code": "Table-as-Search agent code and prompts are public in Marco-Search-Agent; an independently packaged scorer and reference bundle for these 20 tasks were not verified. The repository's DeepWideSearch data/eval directory is separate.",
          "judging_criteria": "Experts verify candidate validity and requested information against all stated constraints. Column-F1 scores entity identification, while Item-Precision scores attribute correctness; fixed quantities, exclusions and dynamic reference unions address open-ended completeness.",
          "reported_results_provenance": "Original authors' Table-as-Search runs are author-reported; no independent reproduction was established here.",
          "reproducibility_and_barriers": "Agent code is public, but a complete bundle of these 20 tasks, reference sets and scoring code was not verified. Live retrieval, commercial interfaces, dynamic reference unions and expert judgments complicate reproduction.",
          "eligibility_date_and_evidence": "2026-02: Table-as-Search (arXiv:2602.06724) introduces the 20-query DeepWide evaluation."
        },
        {
          "name": "DeepWeb-Bench",
          "category": "Deep research benchmark; comprehensive multi-entity evidence and derivation",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2605.21482",
              "supports": "Primary paper, date, task design, metrics, findings and release claim."
            },
            {
              "url": "https://huggingface.co/datasets/deepweb-bench-anon/deepweb-bench",
              "supports": "Released files, cases, result records, executable code and exclusions."
            },
            {
              "url": "https://sixiongxie1001-dot.github.io/deep-research-benchmark2.0",
              "supports": "Official project page."
            }
          ],
          "measures": "Scores retrieval, derivation, reasoning and calibration across 100 cases using entity-by-dimension cells, family scores and source-provenance disclosure levels.",
          "task_data": "Public Hugging Face release: 100 cases, 900 model results/answers/score records, summaries and provenance. Tasks require 6–10 entities across 6–10 dimensions with cross-source evidence.",
          "canonical_url": "https://arxiv.org/abs/2605.21482",
          "evaluation_code": "Actually present in HF `code/`: validation, leaderboard/report rebuilding, rule-prompt grader reruns and an OpenAI-compatible model runner. Aggregation needs no keys; live reruns do.",
          "judging_criteria": "Explicit per-cell rules, provenance levels and cross-source checks determine scores; evaluation is not solely free-form judging.",
          "reported_results_provenance": "Author-reported evaluation of nine frontier models; reported findings concern overall scores and retrieval, derivation and calibration error patterns.",
          "reproducibility_and_barriers": "Data and code are downloadable, but model, grader, search and scrape APIs require keys. Raw tool traces, third-party source snapshots and local MCP/API state are excluded.",
          "eligibility_date_and_evidence": "Submitted 2026-05-20, within cutoff. The primary paper says data, rubrics and evaluation code are publicly released."
        },
        {
          "name": "From Simple QA to Deep Research: A Verifiable Benchmark Constructed through Iterative Task Evolution",
          "category": "Automatically constructed verifiable deep-research benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2608.02163",
              "supports": "Primary date, scope, 500-task design and public code/data claim."
            },
            {
              "url": "https://arxiv.org/html/2608.02163v1",
              "supports": "Official HTML identifies GitHub TaskEvolving repository and method details."
            },
            {
              "url": "https://github.com/chr6192/TaskEvolving.git",
              "supports": "Primary artifact link cited by paper; file-level availability not verified here."
            }
          ],
          "measures": "Evaluates 500 tasks across 31 topics and 10 categories using DAG atomic steps, checkpoints, fact-grounded pointwise rubrics, and model discrimination and stability analyses.",
          "task_data": "Constructed from Wikipedia QA through Explorer–Formalizer–Challenger evolution; each task includes a query, DAG, checkpoints and aligned rubrics. Public availability is claimed by the paper.",
          "canonical_url": "https://arxiv.org/abs/2608.02163",
          "evaluation_code": "Paper links https://github.com/chr6192/TaskEvolving.git and states that implementation, data and results are public. Repository contents were not retrievable in this check, so evaluator files remain unverified.",
          "judging_criteria": "Source-grounded checkpoint and DAG rubrics assign pointwise fact-based scores; the paper describes the method as human-aligned and stable.",
          "reported_results_provenance": "Author-reported experiments demonstrate discrimination across models and query types; no independent results were established.",
          "reproducibility_and_barriers": "Public-release claim is explicit, but construction requires relatively high frontier-model cost. Exact repository data layout and license remain unverified.",
          "eligibility_date_and_evidence": "Submitted 2026-08-03, within cutoff. The primary paper says code and data are publicly available."
        },
        {
          "name": "DeepResearch-9K",
          "category": "challenging multi-hop web research benchmark with agent trajectories",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2603.01152v2",
              "supports": "Primary paper: benchmark construction, 9K/3,974 sizes, difficulty levels, metrics, judge protocol, results, and release links."
            },
            {
              "url": "https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1",
              "supports": "Official code repository: dataset links, hard subset, data format, SFT/RL and inference/evaluation scripts."
            },
            {
              "url": "https://huggingface.co/datasets/artillerywu/DeepResearch-9K",
              "supports": "Official dataset artifact: 9,000/3,974 splits and train/test composition."
            }
          ],
          "measures": "Measures final-answer accuracy, search-tool usage, and trajectory correctness across three difficulty levels.",
          "task_data": "Public 9,000-question Hugging Face release includes difficulty labels, questions, answers, trajectories, and a 3,974-sample hard subset. Data are largely synthetic and teacher-generated.",
          "canonical_url": "https://arxiv.org/abs/2603.01152",
          "evaluation_code": "Public official repository includes training, inference, SFT/RL, environment, and evaluation scripts. The paper’s construction pipeline is released, with DeepSeek-V3 LLM judging.",
          "judging_criteria": "LLM judging scores final-answer correctness; difficulty reflects search-chain complexity and entity obfuscation. Hard items were filtered by incorrect teacher verdicts.",
          "reported_results_provenance": "Authors’ experiments; DeepResearch-R1 is the paired training/agent framework, not an additional benchmark. No independent replication established.",
          "reproducibility_and_barriers": "Data and code are public, but training requires substantial resources and model/API dependencies. Teacher-generated trajectories and LLM judging create assessment dependence.",
          "eligibility_date_and_evidence": "ArXiv paper 2603.01152 was available in March 2026 and identifies SIGIR 2026 publication (July 20–24, 2026), within cutoff. Official repository was created 2026-02-06."
        },
        {
          "name": "Mr.LHDR",
          "category": "Multimodal real-world long-horizon deep-research benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2609.11318",
              "supports": "Primary paper/date, benchmark design, metrics and reported results."
            },
            {
              "url": "https://arxiv.org/html/2609.11318v1",
              "supports": "Official HTML links code repository."
            },
            {
              "url": "https://github.com/minghaoguo20/Mr-LHDR-eval",
              "supports": "Actual public Apache-2.0 repository, requirements, data download/decrypt and run scripts."
            },
            {
              "url": "https://huggingface.co/datasets/Henryeahhh/Mr-LHDR",
              "supports": "Actual dataset schema, source/checklist/image fields and encrypted-looking released records."
            }
          ],
          "measures": "Measures final-answer and dependency-aware conclusion quality using OA, Strict Accuracy, Checklist Score, and Dependency-Aware Checklist Score.",
          "task_data": "Public dataset uses multimodal evidence including images, maps, PDFs, logos, charts, tables, and video frames. Metadata, checklists, sources, and images are released, but some fields are encrypted or canary protected.",
          "canonical_url": "https://arxiv.org/abs/2609.11318",
          "evaluation_code": "Public Apache-2.0 repository includes requirements, decryption, run, and evaluation scripts. Actual execution requires the dataset and model/web-search access.",
          "judging_criteria": "Scores final answers and intermediate conclusions against annotated dependencies using OA, SA, CS, and DACS.",
          "reported_results_provenance": "Author-reported evaluation; strongest system achieved 43.1% OA and 34.3% SA. Removing images reduced DACS by 12.6 points.",
          "reproducibility_and_barriers": "Code and dataset are public, but encrypted metadata and decrypt.py create a material reproduction barrier; raw data are not fully transparent.",
          "eligibility_date_and_evidence": "Submitted 2026-09-10, within 2026-09-26 cutoff."
        },
        {
          "name": "FinSearchComp",
          "category": "financial open-web search and analyst-style reasoning",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2509.13160",
              "supports": "Paper date, 635 questions, task families, metrics, judge, and release links."
            },
            {
              "url": "https://github.com/randomtutu/FinSearchComp",
              "supports": "Observed official repository with data files and eval/eval.py."
            },
            {
              "url": "https://randomtutu.github.io/FinSearchComp/",
              "supports": "Official project page and public-release description."
            }
          ],
          "measures": "Measures binary answer correctness across time-sensitive retrieval, historical lookup, and complex historical investigation.",
          "task_data": "Public 635-question expert-crafted dataset covers Global and Greater China subsets and includes questions, tool templates, answers, and traces.",
          "canonical_url": "https://arxiv.org/abs/2509.13160",
          "evaluation_code": "Actual public evaluator code is available at https://github.com/randomtutu/FinSearchComp, including data/finsearchcomp_data.json, eval/eval.py, and runnable commands.",
          "judging_criteria": "Rubric-guided LLM judging assigns 0/1 correctness, applying numerical tolerance and checking financial conventions and supporting evidence.",
          "reported_results_provenance": "Author-run model comparison; no independent reproduction established.",
          "reproducibility_and_barriers": "Public data and evaluator support reproduction, but live APIs, web changes, and model/tool access remain practical assessment barriers.",
          "eligibility_date_and_evidence": "Submitted 2025-09-16, within cutoff. Official project and repository are public."
        },
        {
          "name": "Finance Agent Benchmark",
          "category": "financial SEC-filing research-agent benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2508.00828",
              "supports": "Primary paper, data splits, judging, metrics, and results."
            },
            {
              "url": "https://github.com/vals-ai/finance-agent",
              "supports": "Observed public harness repository."
            },
            {
              "url": "https://doi.org/10.5281/zenodo.15428823",
              "supports": "Paper-linked harness artifact."
            }
          ],
          "measures": "Measures naive and class-balanced accuracy across nine finance categories, plus execution time and cost.",
          "task_data": "Partially public: 537 expert-authored SEC/EDGAR questions comprise 50 public validation, 150 private validation, and 337 private test items.",
          "canonical_url": "https://arxiv.org/abs/2508.00828",
          "evaluation_code": "Actual MIT-licensed harness and public validation data are available at https://github.com/vals-ai/finance-agent; Zenodo also provides the harness.",
          "judging_criteria": "Rubric-based component grading checks calculations and detects contradictions before assigning correctness.",
          "reported_results_provenance": "Benchmark-author runs; no independent reproduction established.",
          "reproducibility_and_barriers": "Public 50-item validation and code permit partial reproduction; private splits, live filings/search, and paid APIs limit full reproduction (assessment).",
          "eligibility_date_and_evidence": "ArXiv release August 2025, within cutoff."
        },
        {
          "name": "FinRetrieval",
          "category": "financial structured-data retrieval by AI agents",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2603.04403",
              "supports": "Primary paper and release claims, task count, configurations, metrics, and results."
            },
            {
              "url": "https://github.com/daloopa/finretrieval",
              "supports": "Verified public evaluation-code repository and README."
            },
            {
              "url": "https://huggingface.co/datasets/daloopa/finretrieval",
              "supports": "January 2026 dataset card and evaluator link."
            }
          ],
          "measures": "Measures numeric-answer accuracy across 500 questions and 14 model/tool configurations, with tool-call trace analysis.",
          "task_data": "Public release includes 500 questions, 7,000 responses, ground truth, scores, and complete tool-call traces comparing web-only and structured MCP/API access.",
          "canonical_url": "https://arxiv.org/abs/2603.04403",
          "evaluation_code": "Actual public evaluator is available at https://github.com/daloopa/finretrieval; README provides setup and execution commands.",
          "judging_criteria": "Automatic numeric matching compares answers with ground truth; edge cases and fiscal-period conventions receive manual review.",
          "reported_results_provenance": "Daloopa-author evaluation; no independent reproduction established.",
          "reproducibility_and_barriers": "Dataset, evaluator, scores, and traces are public; Daloopa MCP and commercial model/API access remain practical rerun barriers (assessment).",
          "eligibility_date_and_evidence": "January 2026 release; within cutoff."
        },
        {
          "name": "MedBrowseComp",
          "category": "medical live-web/deep-research and biomedical evidence retrieval",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2505.14963",
              "supports": "Canonical paper identity, date, construction, 50/605 evaluation variants, judging and scripts."
            },
            {
              "url": "https://github.com/shan23chen/MedBrowseComp",
              "supports": "Official project repository and actual final50/final121 files/scripts."
            },
            {
              "url": "https://huggingface.co/datasets/AIM-Harvard/MedBrowseComp",
              "supports": "Official dataset artifact with 50 and 605 harmonized datasets."
            }
          ],
          "measures": "Accuracy on MedBrowseComp-50 and MedBrowseComp-605, covering structured extraction and deep research, judged by GPT-4.1-mini with human checking.",
          "task_data": "Public 50-sample and 605-sample benchmarks cover hematology/oncology, PubMed, ClinicalTrials.gov, FDA Orange Book and market data.",
          "canonical_url": "https://arxiv.org/abs/2505.14963",
          "evaluation_code": "Public official repository https://github.com/shan23chen/MedBrowseComp contains final50.csv, final121.csv and processing/evaluation scripts, including process_NCT_predictions.py. The 50/605 counts are verified; the paper describes 121 trials expanded into 605 tasks.",
          "judging_criteria": "GPT-4.1-mini judges answer correctness, with human checking of judge agreement.",
          "reported_results_provenance": "Original authors' commercial-system and Claude computer-use runs; no independent benchmark runner established.",
          "reproducibility_and_barriers": "Partial reproduction is possible with public files and scripts, but live medical sources and proprietary systems create access and drift barriers (assessment).",
          "eligibility_date_and_evidence": "Submitted 2025-05-20, within cutoff; repository and dataset match arXiv 2505.14963."
        },
        {
          "name": "PaSa / AutoScholarQuery and RealScholarQuery",
          "category": "academic literature discovery and comprehensive paper search",
          "evidence": [
            {
              "url": "https://aclanthology.org/2025.acl-long.572/",
              "supports": "Primary ACL paper: datasets, date-constrained scholarly retrieval task, metrics and reported results."
            },
            {
              "url": "https://github.com/bytedance/pasa",
              "supports": "Official code/data repository linked by the primary paper; search result identifies runnable PaSa agent and benchmark support."
            }
          ],
          "measures": "Recall@20, Recall@50, Recall@100 and precision measure relevant-paper retrieval, document selection and crawling.",
          "task_data": "AutoScholarQuery has public 33,551 training, 1,000 development and 1,000 test synthetic queries. RealScholarQuery has 50 manually gathered researcher queries and an annotated 200 query-paper selector set.",
          "canonical_url": "https://aclanthology.org/2025.acl-long.572/",
          "evaluation_code": "Public code and datasets are stated at https://github.com/bytedance/pasa. Repository search results show runnable agent scripts and benchmark support using scholarly retrieval, citation crawling and ar5iv parsing.",
          "judging_criteria": "Ranked relevant-paper retrieval is scored against annotated sets using recall and precision; separate human checks assess query and query–paper relevance.",
          "reported_results_provenance": "Authors' benchmark runs, not independent validation. Reported comparisons are benchmark-author results.",
          "reproducibility_and_barriers": "AutoScholarQuery is comparatively reproducible, while RealScholarQuery may drift with manually collected queries and changing indexes; search/API access and model compute are required (assessment).",
          "eligibility_date_and_evidence": "ACL 2025 primary paper release, within cutoff; official ByteDance repository is linked by the paper and search result."
        },
        {
          "name": "OpenBenchmarks Multi-turn Company Search Benchmark",
          "category": "multi-turn web-search evaluation for comprehensive company-set discovery",
          "evidence": [
            {
              "url": "https://openbenchmarks.com/multi-turn-company-search",
              "supports": "Primary benchmark page: 2026-08-22 publication, subsequent changelog dates, 45-question/375-membership board, public 10-question sample, metrics and locked-vs-public distinction."
            },
            {
              "url": "https://github.com/openbenchmarks-labs/multi-turn-company-search",
              "supports": "Official public runner/judge repository, MIT license, fixed agent, deterministic scorer, artifact contract, credentials and dataset/board caveats."
            },
            {
              "url": "https://huggingface.co/datasets/openbenchmarks/OB-Company-Websearch",
              "supports": "Official advertised endpoint for the separate 10-question public sample; availability should be checked at publication time."
            }
          ],
          "measures": "Precision, recall, F1 and exact-set accuracy score company-set discovery; median latency and cost measure efficiency across search-only and search-plus-fetch settings.",
          "task_data": "Full board gold is locked: 45 hand-labelled questions with 375 canonical memberships. A separate public 10-question search-only sample with frozen reference companies is available via Hugging Face.",
          "canonical_url": "https://openbenchmarks.com/multi-turn-company-search",
          "evaluation_code": "Public MIT-licensed GitHub runner and deterministic offline judge validate schemas, receipts, canonical matching and aggregation. Running agents requires provider credentials; saved-artifact judging is offline.",
          "judging_criteria": "Canonical set comparison counts true positives, false positives and false negatives for precision, recall, F1 and exact-set accuracy. Three trials are aggregated; SD measures trial variability, not confidence.",
          "reported_results_provenance": "OpenBenchmarks editorial-board/vendor runs; no independent results established. Board scores use the locked 45-question set, while the public sample is not score-comparable.",
          "reproducibility_and_barriers": "Partial reproduction is possible with the public runner, judge and 10-question sample, but locked gold, live providers, costs and web drift remain barriers (assessment).",
          "eligibility_date_and_evidence": "Official page records first complete benchmark publication on 2026-08-22, additions on 2026-08-26 and updates on 2026-09-15; all fall within the cutoff."
        },
        {
          "name": "MM-BrowseComp",
          "category": "multimodal deep web browsing benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2508.13186",
              "supports": "Primary paper/date and multimodal browsing design; v1 task count and checklist concept."
            },
            {
              "url": "https://github.com/MMBrowseComp/MM-BrowseComp",
              "supports": "Official release history, current 400-question update, encrypted data and concrete decrypt/generate/evaluate commands."
            }
          ],
          "measures": "Final-answer accuracy measures correctness; checklist analysis measures multimodal dependencies and reasoning paths beyond aggregate scores.",
          "task_data": "Public encrypted JSONL includes the current 400-question update at data/MMBrowseComp_400.jsonl. Prompts and evidence may include images and videos; canary/decryption protects questions and answers.",
          "canonical_url": "https://github.com/MMBrowseComp/MM-BrowseComp",
          "evaluation_code": "Public src/decrypt.py, src/gen_answer.py and src/eval.py implement decryption, answer generation and LLM judging. The repository includes released data and evaluation scripts.",
          "judging_criteria": "An LLM judges reference-answer correctness, while verified per-question checklists diagnose dependency and reasoning paths.",
          "reported_results_provenance": "Paper/author-reported model evaluations; no independent result established here.",
          "reproducibility_and_barriers": "Partial reproduction is possible with public code/data, but encrypted contents, canary/decryption, API-backed models, live web access and configuration create barriers (assessment).",
          "eligibility_date_and_evidence": "ArXiv v1 2025-08-14; official repository says full codebase released 2025-08-20 and dataset expanded to 400 questions 2026-01-02. Version drift remains: v1 describes 224/244 questions, current release 400."
        },
        {
          "name": "MMSearch-Plus",
          "category": "provenance-aware multimodal browsing/search benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2508.21475",
              "supports": "Primary paper date, 311 tasks, curation/evaluation design and reported results."
            },
            {
              "url": "https://github.com/mmsearch-plus/MMSearch-Plus",
              "supports": "Official release dates, dataset link, decryption usage, framework/evaluation and annotation availability."
            },
            {
              "url": "https://huggingface.co/datasets/Cie1/MMSearch-Plus",
              "supports": "Public dataset artifact linked by the project."
            }
          ],
          "measures": "Answer accuracy measures correctness on 311 tasks; bounding-box, cropping and provenance analyses measure localized visual search and evidence-chain behavior.",
          "task_data": "Public encrypted dataset includes questions/images, ground truth, alternatives, metadata and Set-of-Mark annotations. Tasks require localized visual cues, spatial-temporal reasoning, iterative retrieval and cross-validation.",
          "canonical_url": "https://github.com/mmsearch-plus/MMSearch-Plus",
          "evaluation_code": "Public official repository provides the agentic rollout framework, evaluation script and Set-of-Mark annotations; Hugging Face data are linked. External search infrastructure is not claimed to be bundled.",
          "judging_criteria": "Answers are scored for accuracy, while visual localization/cropping and provenance-aware retrieval analyses assess multimodal evidence chains rather than text-only shortcuts.",
          "reported_results_provenance": "Author-run paper/official leaderboard results; independent reproduction not verified.",
          "reproducibility_and_barriers": "Partial reproduction requires public data/code plus decryption, model/API and live-search dependencies; changing image sources and retrieval noise may affect results (assessment).",
          "eligibility_date_and_evidence": "ArXiv released 2025-08-29; official README records all data samples released to Hugging Face 2025-09-26, within cutoff."
        },
        {
          "name": "BrowseComp-V³ (BrowseComp-V3)",
          "category": "visual, vertical, verifiable multimodal deep-search benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2602.12876",
              "supports": "Primary benchmark description, 300 tasks, process evaluation and reported results."
            },
            {
              "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3",
              "supports": "Official repository contents: dataset download/decryption, assets, evaluator, baseline runner and license documentation."
            },
            {
              "url": "https://halcyon-zhang.github.io/BrowseComp-V3/",
              "supports": "Official project page and artifact links, including dataset/repository."
            }
          ],
          "measures": "Measures final-answer success and process-level adherence using expert-validated intermediate subgoals and search trajectories.",
          "task_data": "Partial/public: encrypted downloadable data contain 300 questions across 24 subdomains, visual assets, evidence, gold trajectories and subgoals; repository scripts decrypt JSON and images.",
          "canonical_url": "https://github.com/Halcyon-Zhang/BrowseComp-V3",
          "evaluation_code": "Actual public repository code includes dataset download/decryption, rollout evaluation, score summarization, an OmniSeeker runner, documentation and smoke tests.",
          "judging_criteria": "Scores final-answer correctness and success alongside process and subgoal adherence for cross-modal, multi-hop evidence integration.",
          "reported_results_provenance": "Author-reported paper/project evaluations include OmniSeeker and MLLM comparisons; independent results are not verified.",
          "reproducibility_and_barriers": "Public GitHub/Hugging Face artifacts and CC BY 4.0 claim; encrypted samples, key, live search, judge model and APIs remain dependencies. Assessment: web volatility and closed baselines limit exact reproduction.",
          "eligibility_date_and_evidence": "2026-02: arXiv:2602.12876 and official project/repository. Unicode V³ and ASCII V3 designate the same benchmark."
        },
        {
          "name": "ScholarQuest",
          "category": "Academic literature/paper-search benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2606.20235",
              "supports": "Paper identity, 2026-06-18 submission, benchmark scope, 1,111 queries, four intents, metrics and headline results."
            },
            {
              "url": "https://github.com/pty12345/ScholarQuest",
              "supports": "Public benchmark dataset, code, construction pipeline, ScholarBase/Lewen backend and dataset structure."
            },
            {
              "url": "https://arxiv.org/html/2606.20235",
              "supports": "Detailed benchmark construction, corpus/retrieval design and evaluation methodology."
            }
          ],
          "measures": "Measures retrieval recall at 25, 100 and all results, plus search efficiency, tool use and robustness to intent and answer-set size.",
          "task_data": "Public: 1,111 queries from 1,000+ CS topic seeds use four intents, arXiv answer sets of 5–200 papers and a million-scale ScholarBase corpus with metadata and citations.",
          "canonical_url": "https://arxiv.org/abs/2606.20235",
          "evaluation_code": "Actual public repository includes the benchmark dataset, construction pipeline, analysis scripts and ScholarBase/Lewen search backend.",
          "judging_criteria": "Scores retrieval recall against answer_arxiv_ids after arXiv-ID normalization; process statistics cover rounds, calls, candidates and recall efficiency.",
          "reported_results_provenance": "Paper authors report PaperScout and hybrid baselines; results are author-reported, with no independent validation established here.",
          "reproducibility_and_barriers": "Public code/data; reproduction requires deploying ScholarBase/Lewen and potentially using external search APIs or models (assessment).",
          "eligibility_date_and_evidence": "Submitted 2026-06-18 (within window); public code/data repository linked by paper."
        },
        {
          "name": "Sage: Benchmarking and Improving Retrieval for Deep Research Agents",
          "category": "Scientific literature retrieval benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2602.05975",
              "supports": "Benchmark definition, 1,200 queries, four domains, 200,000-paper corpus and reported findings."
            },
            {
              "url": "https://arxiv.org/html/2602.05975v2",
              "supports": "Dataset composition, corpus-search setup, metrics/results and corpus-scaling method."
            },
            {
              "url": "https://github.com/HughieHu/Sage",
              "supports": "Public implementation/artifact linked as the SAGE repository."
            }
          ],
          "measures": "Reasoning-intensive scientific paper discovery: exact-match target-paper retrieval and relevance-weighted coverage of open-ended literature queries.",
          "task_data": "Public repository contains 600 short-form and 600 open-ended queries, with paper IDs/titles and relevance-tier references. The paper describes four domain-specific 50,000-paper corpora; a complete corpus bundle was not verified.",
          "canonical_url": "https://arxiv.org/abs/2602.05975",
          "evaluation_code": "Public question/reference JSON files and metric definitions are verified. The inspected repository README does not establish a turnkey evaluator or full agent/retriever implementation.",
          "judging_criteria": "Short-form exact match checks whether the target paper appears in answer text or citations. Open-ended weighted recall gives seed papers relevance 2, shared-reference papers 1 and other papers 0.",
          "reported_results_provenance": "Authors evaluate six deep-research agents and DR Tulu-backed retrievers; reported findings are author-generated, not independently validated here.",
          "reproducibility_and_barriers": "Query/reference data are public; reproducing agent experiments requires the corresponding corpus, retrieval indexes and model/API setup. Corpus augmentation adds processing cost.",
          "eligibility_date_and_evidence": "Submitted 2026-02-06 (within window); public GitHub repository linked from paper."
        },
        {
          "name": "AstaBench (literature-search and research-synthesis components)",
          "category": "Core subset — scientific literature discovery/synthesis; broader science suite is adjacent",
          "evidence": [
            {
              "url": "https://allenai.org/asta/bench",
              "supports": "Official benchmark page listing literature components, metrics and evaluation framework."
            },
            {
              "url": "https://github.com/allenai/asta-bench",
              "supports": "Public code, task names/datasets, standardized tools and date/corpus restrictions."
            },
            {
              "url": "https://arxiv.org/abs/2510.21652",
              "supports": "Benchmark paper date, suite scope (2,400+ problems), environment and agent evaluation design."
            },
            {
              "url": "https://allenai.org/blog/astabench",
              "supports": "Official release date and reported literature-component results."
            }
          ],
          "measures": "Measures paper finding, literature retrieval and QA, long-form review answers, literature-review tables, quality, cost and tool-controlled agent behavior.",
          "task_data": "Public: 11 benchmarks and 2,400+ problems cover literature, coding, data analysis and discovery; literature components include PaperFindingBench, ScholarQA-CS2, LitQA2 and ArxivDIGESTables-Clean.",
          "canonical_url": "https://allenai.org/asta/bench",
          "evaluation_code": "Actual public GitHub suite provides task datasets, interfaces and standardized tools; Asta Scientific Corpus supports date- and corpus-restricted search and retrieval.",
          "judging_criteria": "Scores task-specific answers or retrieval; ScholarQA-CS2 adds coverage and citation precision, while logs record cost, tools and traces.",
          "reported_results_provenance": "Ai2 authors report 57-agent evaluations; announcement reports Asta Scholar QA/Elicit/SciSpace and Asta Paper Finder results. Author/vendor results, not independently validated.",
          "reproducibility_and_barriers": "Open benchmark/framework and baseline agents; reproduction depends on Asta Environment or corpus snapshots and compatible agent-evaluation tooling (assessment).",
          "eligibility_date_and_evidence": "Public Ai2 announcement dated 2025-08-26 and benchmark paper submitted 2025-10-24; within window."
        },
        {
          "name": "LiveDRBench (Microsoft), from “Characterizing Deep Research: A Benchmark and Formal Definition”",
          "category": "open-web deep-research claim discovery; includes entity/list retrieval and evidence-grounded synthesis",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2508.04183",
              "supports": "Formal definition, open-web scope, 100-task domains, claim representation, metrics, categories and author results."
            },
            {
              "url": "https://github.com/microsoft/LiveDRBench",
              "supports": "Public repository date, 100-task dataset description, Hugging Face loading, evaluation script/instructions and licenses."
            }
          ],
          "measures": "Measures claim-level precision, recall and F1, including recursive claim/subclaim correctness; also records sources, branching and backtracking.",
          "task_data": "Public: 100 live open-web tasks cover science and world events, with prompts, output formats, ground-truth JSON claims and references; Hugging Face provides the dataset.",
          "canonical_url": "https://github.com/microsoft/LiveDRBench",
          "evaluation_code": "Actual public MIT repository includes src/evaluate.py and instructions; GPT-4o via OpenAI API judges claim agreement before computing information-retrieval metrics.",
          "judging_criteria": "Scores correctness and completeness of substantive claims; GPT-4o maps predictions to ground-truth claims, and unsupported incorrect subclaims receive no credit.",
          "reported_results_provenance": "Microsoft authors report runs involving OpenAI, Perplexity, Google/Gemini and an open-source agent; independent reproduction is not established here.",
          "reproducibility_and_barriers": "Public code/data and a stated refresh plan support reproduction, but live-web changes and GPT/API dependence affect repeatability. It is explicitly open-web, not enterprise search.",
          "eligibility_date_and_evidence": "Public repository created 2025-07-25; paper arXiv:2508.04183 is 2025. Data were collected May–June 2025, within the requested window."
        },
        {
          "name": "DEER: A Benchmark for Evaluating Deep Research Agents on Expert Report Generation (LG AI Research)",
          "category": "long-form expert report quality, evidence grounding and report-wide fact verification",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2512.17776v2",
              "supports": "Task count/domains, taxonomy, 101 rubrics, expert guidance, fact verification and judging architecture."
            },
            {
              "url": "https://github.com/hanjanghoon/DEER",
              "supports": "Official code, repository date, dataset location, access procedure, licenses and redistribution restrictions."
            },
            {
              "url": "https://huggingface.co/datasets/LG-AI-Research/DEER-Deep-Research-Benchmark",
              "supports": "Official dataset artifact and custom-license/access status."
            }
          ],
          "measures": "Measures report fulfillment, analytical soundness, coherence, style and ethics through 101 rubric items, plus claim factuality, citation support, evidence quality and sufficiency.",
          "task_data": "Public benchmark artifact, but access is gated by a password-protected archive; it contains 50 expert-report tasks across 13 domains derived from Humanity’s Last Exam and internal queries.",
          "canonical_url": "https://github.com/hanjanghoon/DEER",
          "evaluation_code": "Official MIT-licensed evaluation code is public; dataset access is gated under a custom license permitting non-commercial research but prohibiting redistribution, mirroring and public posting.",
          "judging_criteria": "LLM judges apply fixed rubric items and expert guidance; a fact-verification module extracts claims and citations, then checks external evidence through web search.",
          "reported_results_provenance": "Reported human-correlation and system-comparison results are LG AI Research authors’ experiments; independent validation was not established here.",
          "reproducibility_and_barriers": "Gated data, web-search verification, LLM judging, licensing restrictions and anti-contamination requirements materially limit fully frictionless reproduction.",
          "eligibility_date_and_evidence": "Paper arXiv:2512.17776 first released 2025-12-19; official repository created 2026-02-03. Both fall within the requested window."
        },
        {
          "name": "PaSaMaster-Bench",
          "category": "multidisciplinary scientific literature retrieval / comprehensive paper-set discovery",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2605.14306",
              "supports": "2026 paper identity, benchmark definition, 244 tasks, 38 disciplines, construction, metrics and author results."
            },
            {
              "url": "https://arxiv.org/abs/2605.14306",
              "supports": "Benchmark protocol, checklist-based ground truth, top-20 metrics and reliability measures."
            },
            {
              "url": "https://github.com/sjtu-sai-agents/PaSaMaster",
              "supports": "Official system repository and release link; benchmark data/evaluator availability was not verified."
            }
          ],
          "measures": "Measures top-20 recall, precision, F1 and NDCG, plus source-hallucination rate and token cost; the paper reports 244 expert-curated tasks across 38 disciplines.",
          "task_data": "Availability not established: tasks contain multi-constraint literature intents, target paper sets and expert checklist annotations, but a public task/answer bundle was not confirmed.",
          "canonical_url": "https://arxiv.org/abs/2605.14306",
          "evaluation_code": "Public PaSaMaster system and retrieval code are available; benchmark-specific scorer and data files were not verified, so evaluation release status remains uncertain.",
          "judging_criteria": "Expert checklists determine whether retrieved papers satisfy the full intent; top-20 set and ranking metrics compare results with target papers, while source checks support hallucination scoring.",
          "reported_results_provenance": "Reported results are the paper authors’ runs; no independent reproduction was established.",
          "reproducibility_and_barriers": "Assessment: live scholarly sources, proprietary model/search comparisons and a potentially unreleased benchmark bundle impede exact reproduction, although system code is public.",
          "eligibility_date_and_evidence": "2026-05: arXiv:2605.14306 introduces PaSaMaster-Bench, distinct from PaSa’s earlier AutoScholarQuery/RealScholarQuery datasets."
        },
        {
          "name": "AutoResearchBench",
          "category": "scientific literature deep identification and comprehensive set discovery",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2604.25256",
              "supports": "Paper date, task paradigms, 1,000-query composition, construction, metrics and reported evaluation."
            },
            {
              "url": "https://github.com/CherYou/AutoResearchBench",
              "supports": "Official inference/evaluation code, decryptor, task mapping and Hugging Face data instructions."
            },
            {
              "url": "https://cheryou.github.io/autoresearchbench.github.io/",
              "supports": "Official project page, split, schema, answers, license and public evaluation workflow."
            },
            {
              "url": "https://huggingface.co/datasets/Lk123/AutoResearchBench",
              "supports": "Officially referenced host for the obfuscated benchmark bundle."
            }
          ],
          "measures": "Deep research uses exact-answer accuracy; wide research uses set IoU for coverage and precision. The benchmark contains 1,000 problems: 600 deep and 400 wide.",
          "task_data": "Public obfuscated JSONL bundle on Hugging Face, decrypted locally; it covers eight CS areas and contains human-verified deep and wide literature-search tasks.",
          "canonical_url": "https://arxiv.org/abs/2604.25256",
          "evaluation_code": "Public GitHub code includes inference, academic/web search tools, prompts, utilities, decryption scripts and separate deep- and wide-search evaluators.",
          "judging_criteria": "Deep answers receive exact-match scoring; wide answers are scored by IoU against gold paper sets using a standardized ReAct agent and DeepXiv search setup.",
          "reported_results_provenance": "Results are the original authors’ baseline runs across more than 10 models and agents; independent reproduction was not established.",
          "reproducibility_and_barriers": "Public code, data and evaluator support reproduction; decryption, model/API credentials, search access, token cost and live-search drift remain practical barriers.",
          "eligibility_date_and_evidence": "arXiv 2604.25256 (2026) falls within the window; official project resources publicly release code and benchmark data."
        },
        {
          "name": "DRBENCHER (Deep Research Benchmarker)",
          "category": "entity identification + property retrieval + quantitative computation for web research agents",
          "evidence": [
            {
              "url": "https://research.ibm.com/publications/drbencher-can-your-agent-identify-the-entity-retrieve-its-properties-and-do-the-math",
              "supports": "IBM Research publication page, benchmark purpose, domains, validity, accuracy and release links."
            },
            {
              "url": "https://arxiv.org/html/2604.09251v3",
              "supports": "Answer-first generation, CCI, sources, validation, metrics, human/model results and error analysis."
            },
            {
              "url": "https://github.com/IBM/DrBencher",
              "supports": "Primary code/data repository, released QA files, schema, license and reproducible output fields."
            }
          ],
          "measures": "Measures answer accuracy, entity identification, validity, human quality and semantic diversity for multi-hop entity, property-retrieval and calculation tasks; difficulty is summarized by CCI.",
          "task_data": "Public JSONL tasks are available. The paper describes 268 human-validated questions, while repository documentation lists a 255-question main evaluation set; pin versions rather than treating these counts as interchangeable.",
          "canonical_url": "https://arxiv.org/abs/2604.09251",
          "evaluation_code": "Public MIT-licensed IBM/DrBencher repository includes pipeline, schemas, released JSONL data, decryption/evaluation utilities and computation-based gold-answer checking.",
          "judging_criteria": "Programmatic checks require reproducible calculations, supported clues, no entity leakage and unambiguous questions; answers use domain-specific tolerances, while humans assess validity.",
          "reported_results_provenance": "Results are IBM authors’ human annotations and six-model evaluations, not independent reproductions; the paper identifies property retrieval as the dominant failure mode.",
          "reproducibility_and_barriers": "Public code, schema and data support reproduction; live Wikidata/Wikipedia and domain APIs, changing values, model/API access and stale data remain barriers.",
          "eligibility_date_and_evidence": "arXiv first posted 2026-04-10 (2604.09251; v3 rendered 2026-08-09), within cutoff; IBM Research and the paper link the public implementation."
        },
        {
          "name": "Reka Research-Eval (including the ndurner evaluator extension)",
          "category": "search-augmented web research / grounded multi-hop question answering (static QA-style task suite, not comprehensive list-building)",
          "evidence": [
            {
              "url": "https://reka.ai/news/introducing-research-eval-a-benchmark-for-search-augmented-llms",
              "supports": "Dated release, 374 questions, checklist judging, six-stage annotation process, and author-reported results."
            },
            {
              "url": "https://github.com/reka-ai/research-eval",
              "supports": "Upstream repository establishes the original task/evaluator provenance and distinguishes the fork from the source suite."
            },
            {
              "url": "https://github.com/ndurner/web-research-eval",
              "supports": "Fork README identifies the inherited benchmark, added APIs, workflow, five-run leaderboard, Durner-marked rows, and reproduction script."
            },
            {
              "url": "https://ndurner.github.io/reka-websearch-benchmark",
              "supports": "Author's extension page explains that Reka released the 374-question benchmark and that the fork adds OpenAI Responses API and Exa Answers comparisons."
            }
          ],
          "measures": "Measures checklist-based answer accuracy for grounded multi-hop web questions, with aggregate mean accuracy and cost per 1,000 requests; it does not measure exhaustive recall or report quality.",
          "task_data": "Public encrypted dataset of 374 questions with correctness checklists, constructed through generation, annotation, consensus refinement and filtering. Answering may use live web sources.",
          "canonical_url": "https://github.com/reka-ai/research-eval",
          "evaluation_code": "Public Reka dataset and generation, scoring, and analysis scripts; ndurner/web-research-eval is a fork extending provider/model support on the same task suite, not a new benchmark.",
          "judging_criteria": "An LLM judge checks each answer against its question-specific checklist, and aggregate accuracy is the resulting score; source quality and citation entailment are not scored.",
          "reported_results_provenance": "Reka launch scores are benchmark-author/vendor results. Nils Durner reports separate third-party extension runs; organizational or financial independence was not established.",
          "reproducibility_and_barriers": "Requires provider credentials, live search and an LLM judge. The Modified MIT license prohibits redistributing decrypted data. Pin fork, model, provider and run settings; author and third-party runs are not automatically comparable.",
          "eligibility_date_and_evidence": "Public release announced by Reka on 2025-08-28; repository created 2025-08-27, within the 2025-01-01–2026-09-26 window."
        },
        {
          "name": "VibeSearchBench: Benchmarking Long-horizon Proactive Search in the Wild",
          "category": "interactive web research/discovery with evolving intent, multi-turn proactive search, and structured evidence enrichment",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2605.27882",
              "supports": "Primary paper: date, 200 bilingual tasks/20 domains, annotation and dual review, simulator, graph-matching evaluator, metrics, experimental setup, and reported results."
            },
            {
              "url": "https://github.com/VibeBench/VibeSearchBench",
              "supports": "Official implementation: task files, agent/tool code, eval/grader/evaluator modules, run scripts, dataset fields, and metric definitions."
            },
            {
              "url": "https://vibebench.github.io/VibeSearchBench.github.io/index.html",
              "supports": "Official project page: proactive-search framing, professional/daily task subsets, multi-turn tools, persona simulator, and graph-F1 evaluation."
            }
          ],
          "measures": "Measures node and knowledge-graph triplet precision, recall, and F1 under average-at-N and best-at-N aggregation, evaluating discovery and structured enrichment.",
          "task_data": "Public dataset of 200 manually curated bilingual tasks across 20 domains: 100 professional and 100 daily scenarios, evenly split between Chinese and English, with expert-annotated ground-truth graphs.",
          "canonical_url": "https://github.com/VibeBench/VibeSearchBench",
          "evaluation_code": "Official repository includes public task JSON, agent implementations, search/visit/Python toolkits, user simulation, evaluation and grading modules, compatible judges, and inference/evaluation scripts.",
          "judging_criteria": "Two-phase LLM graph matching handles aliases and translations, then semantic relations; recall allows direct, subsuming, collective, or compositional coverage, while precision counts covered predicted triples.",
          "reported_results_provenance": "Paper authors evaluate seven frontier models with ReAct and OpenClaw and report F1 and ablations; no independent reproduction was established.",
          "reproducibility_and_barriers": "Tasks, code, and evaluator are public, but execution requires LLM/search credentials and substantial live multi-turn inference; provider drift and public ground truth create assessment barriers.",
          "eligibility_date_and_evidence": "arXiv paper published/submitted May 2026 (arXiv:2605.27882); GitHub repository public May 20, 2026, within the window."
        },
        {
          "name": "DeepWideSearch",
          "category": "deep-and-wide agentic information-seeking benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2510.20168",
              "supports": "Canonical paper, 220 count, 15 domains, split composition, metrics, and results."
            },
            {
              "url": "https://github.com/AIDC-AI/Marco-Search-Agent",
              "supports": "Repository tree showing DeepWideSearch data/eval/scripts."
            },
            {
              "url": "https://huggingface.co/datasets/AIDC-AI/DeepWideSearch",
              "supports": "Public 220-question dataset and reproduction instructions."
            }
          ],
          "measures": "Measures structured-table retrieval using exact task success, row/item F1, and core-entity accuracy across repeated runs.",
          "task_data": "Public dataset of 220 bilingual English/Chinese questions across 15 domains: 85 Deep2Wide and 135 Wide2Deep, averaging 414.10 information units and 4.21 reasoning depth.",
          "canonical_url": "https://arxiv.org/abs/2510.20168",
          "evaluation_code": "Public data and evaluation code are available in https://github.com/AIDC-AI/Marco-Search-Agent under Marco-DeepResearch-Family/DeepWideSearch/data, eval, and scripts; a public HF artifact also exists.",
          "judging_criteria": "Human-verified table ground truth is used; exact success requires all rows, columns, and values, while row/item F1 and core-entity accuracy provide partial scores.",
          "reported_results_provenance": "Authors report four-run system comparisons; no independent reproduction was established.",
          "reproducibility_and_barriers": "Public data and evaluation scripts support reproduction, but live retrieval, multi-run cost, and human annotation remain assessment barriers.",
          "eligibility_date_and_evidence": "Submitted/published October 2025, within cutoff; this is the canonical DeepWideSearch paper, not the later Table-as-Search paper."
        },
        {
          "name": "BrowseComp-VL",
          "category": "multimodal web discovery (introduced with WebWatcher)",
          "evidence": [
            {
              "url": "https://arxiv.org/pdf/2508.05748",
              "supports": "Primary paper introduces BrowseComp-VL, task construction, level counts and original comparisons."
            },
            {
              "url": "https://github.com/alibaba-nlp/deepresearch/blob/main/WebAgent/WebWatcher/README.md",
              "supports": "Official benchmark options, image-download caveat and inference/evaluation instructions."
            }
          ],
          "measures": "Measures multimodal, multi-hop web information seeking through final-answer accuracy/Pass@1, including identification of obfuscated entities from images and textual clues.",
          "task_data": "Paper describes 199 level-1 and 200 level-2 image/question pairs. Task-bundle availability is partial: the repository expects JSONL files and separately downloaded images.",
          "canonical_url": "https://arxiv.org/abs/2508.05748",
          "evaluation_code": "Public WebWatcher repository documents benchmark inference and evaluation scripts; it provides an agent harness rather than a separately packaged benchmark evaluator.",
          "judging_criteria": "Scores reference-answer correctness separately for the two difficulty levels; the exact judging configuration was not verified.",
          "reported_results_provenance": "Benchmark-author/Alibaba WebWatcher experiments; distinct from the separately authored MM-BrowseComp and BrowseComp-V³. No independent rerun was established.",
          "reproducibility_and_barriers": "Assessment barriers include image or OSS download failures, required local evaluation data, live search/image retrieval, model access, and agent dependencies.",
          "eligibility_date_and_evidence": "2025-08: WebWatcher paper introduces BrowseComp-VL; official repository documents benchmark-specific inference/evaluation."
        },
        {
          "name": "Deep Research Comparator",
          "category": "human evaluation framework for research reports and intermediate steps",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2507.05495v1",
              "supports": "Evaluation design, dated paper, study size and data-release promise."
            },
            {
              "url": "https://github.com/cxcscmu/Deep-Research-Comparator",
              "supports": "Actual public platform repository, creation date, dependencies and setup."
            }
          ],
          "measures": "Measures side-by-side report preferences, intermediate-step quality, and text-span feedback; it is not a fixed answer-key benchmark.",
          "task_data": "Availability is partial: the paper reports 176 user queries, 17 annotators, and three agents, but public release of the collected annotation corpus was not established and is promised.",
          "canonical_url": "https://github.com/cxcscmu/Deep-Research-Comparator",
          "evaluation_code": "Public frontend, backend, agent-service integration, and configuration instructions establish platform-code availability, not release of the collected study data.",
          "judging_criteria": "Human pairwise preferences produce outcome rankings, while up/down votes on intermediate steps and report spans provide process-level feedback.",
          "reported_results_provenance": "Authors’ proof-of-concept user study; no independent replication was established. Simple Deepresearch is the accompanying agent scaffold, not a separate benchmark.",
          "reproducibility_and_barriers": "Requires human annotators, Python 3.12, Node.js 18+, PostgreSQL, and agent API/search credentials; live sources and rater variation prevent deterministic replay.",
          "eligibility_date_and_evidence": "2025-07: paper arXiv:2507.05495; official MIT repository created 2025-07-05."
        },
        {
          "name": "Cross-Lingual BrowseComp-Plus (XBCP)",
          "category": "controlled deep-research retrieval and evidence-grounded answering; multilingual extension distinct from BrowseComp-Plus",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2606.15345",
              "supports": "Paper: date, benchmark distinction, evidence construction, languages, task counts, metrics, oracle analysis and results."
            },
            {
              "url": "https://github.com/paddler2022/XBCP",
              "supports": "Repository: downloads, indexing, agent, oracle, evaluation scripts and artifact layout."
            },
            {
              "url": "https://huggingface.co/datasets/UTokyo-Yokoya-Lab/XBCP",
              "supports": "Dataset location; viewer reports schema-generation error, not absent files."
            }
          ],
          "measures": "Measures end-to-end answer accuracy, gold-evidence recall, search-call cost, calibration, citation coverage/precision/recall, oracle-retrieval accuracy, and language/retrieval gaps.",
          "task_data": "Public availability is established through the repository and linked HF artifacts. It preserves English questions and answers while translating evidence documents into 12 languages, with cross-lingual and multilingual configurations.",
          "canonical_url": "https://arxiv.org/abs/2606.15345",
          "evaluation_code": "Public MIT repository includes preparation, translation, indexing, agent-running, oracle, LLM-judge, and per-language evaluation scripts; this is actual released code, using GPT-5.4 through OpenRouter.",
          "judging_criteria": "LLM judges final-answer correctness; evidence recall is computed against gold documents, citation metrics assess attribution, and oracle settings isolate retrieval from language-mismatch integration.",
          "reported_results_provenance": "Authors report runs across four agents and several retrievers, including translated-evidence accuracy drops and weaker evidence and citation performance; no independent validation identified.",
          "reproducibility_and_barriers": "Public code claims a complete pipeline, but reproduction requires decrypted original BrowseComp-Plus material, HF downloads, large indexes, APIs, and live models. The HF viewer reports a schema error.",
          "eligibility_date_and_evidence": "Submitted June 13, 2026 (v2 June 17, 2026), within cutoff; official repository created June 13, 2026."
        },
        {
          "name": "K-BrowseComp: A Web Browsing Agent Benchmark Grounded in Korean Contexts",
          "category": "Korean difficult web-discovery and short-answer browsing benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/html/2606.02404",
              "supports": "Paper: date, 400-item design, validation, metrics, protocol and author results."
            },
            {
              "url": "https://github.com/prometheus-eval/K-BrowseComp",
              "supports": "Official code: evaluation, generation, loading, fallback data and trajectory fields."
            },
            {
              "url": "https://huggingface.co/datasets/prometheus-eval/k-browsecomp",
              "supports": "Public dataset: splits, sizes, source metadata, trajectories and license."
            }
          ],
          "measures": "Measures Pass@1 answer accuracy and calibration on 300 verified items, separate synthetic-item accuracy, and trajectory/search-call behavior for multi-hop or branching questions.",
          "task_data": "Public dataset availability is established. It contains 400 items: 300 Korean-speaker-validated handcrafted problems and 100 synthetic diagnostic items, with answers, URLs, trajectories, checklists and metadata.",
          "canonical_url": "https://arxiv.org/abs/2606.02404",
          "evaluation_code": "Public official repository includes generation and runtime evaluation code, search-evals harness, dataset loading and fallback JSONL; it uses Perplexity Search API and GPT-5.4-mini extraction.",
          "judging_criteria": "Scores single-run Pass@1 against short gold answers and reports calibration; synthetic results remain separate, while trajectories and checklists support diagnostic analysis.",
          "reported_results_provenance": "Authors report single-run evaluations using a common Perplexity pipeline and diagnose termination, trajectory, candidate-management and constraint-tracking failures; no independent results identified.",
          "reproducibility_and_barriers": "Public MIT code and data are documented, but API/model access and changing Korean web results are required. Gitignored seed material limits exact reconstruction of synthetic generation.",
          "eligibility_date_and_evidence": "ArXiv v1 dated June 1, 2026, within cutoff; repository published May 31, 2026 and dataset card is public."
        },
        {
          "name": "DR-Arena: an Automated Evaluation Framework for Deep Research Agents",
          "category": "Automated dynamic deep-research evaluation of depth and breadth",
          "evidence": [
            {
              "url": "https://aclanthology.org/2026.acl-long.1249/",
              "supports": "ACL publication date, framework, measures and reported correlation."
            },
            {
              "url": "https://arxiv.org/html/2601.10504v1",
              "supports": "Release date, pipeline, Examiner judging and evaluation details."
            },
            {
              "url": "https://github.com/iNLP-Lab/DR-Arena",
              "supports": "Public implementation and retained 30-tree dataset location."
            }
          ],
          "measures": "Measures reasoning depth through tree deduction, coverage breadth through aggregation, and pairwise win/Elo; the paper also reports correlation with LMSYS Search Arena.",
          "task_data": "Public retained dataset availability is established: 30 evaluation trees are included in the repository. New trees can be generated by crawling current web trends.",
          "canonical_url": "https://aclanthology.org/2026.acl-long.1249/",
          "evaluation_code": "Public GitHub repository includes arena logic, tree generation/crawling, Examiner question generation and judging, scoring/Elo, and the retained 30-tree dataset.",
          "judging_criteria": "An automated Examiner builds source-grounded rubrics and judges answers and reports for evidence-based correctness; adaptive evaluation escalates depth or breadth, with reported human validation.",
          "reported_results_provenance": "Six-model results and LMSYS correlation are original author and benchmark runs; LMSYS Search Arena supplies the external human-comparison reference.",
          "reproducibility_and_barriers": "The retained trees support fixed comparisons, whereas live-tree generation is time-sensitive. Reproduction requires model, search and API access, and results may change with web or Examiner updates.",
          "eligibility_date_and_evidence": "Public arXiv release January 15, 2026; ACL publication July 2–7, 2026, within cutoff."
        },
        {
          "name": "Personalized Deep Research Bench (PDR-Bench), in Towards Personalized Deep Research: Benchmarks and Evaluations",
          "category": "Personalized deep-research report benchmark",
          "evidence": [
            {
              "url": "https://arxiv.org/abs/2509.25106",
              "supports": "Submission and revision dates and benchmark scope."
            },
            {
              "url": "https://github.com/OPPO-PersonalAI/PersonalizedDeepResearchBench",
              "supports": "Task/profile counts, data paths, scripts, evaluators and API requirements."
            },
            {
              "url": "https://huggingface.co/datasets/PersonalAILab/PersonalizedDeepResearchBench",
              "supports": "Public dataset release artifact."
            }
          ],
          "measures": "Measures personalization alignment across goal, content, presentation and actionability; content quality through depth, insight, coherence and clarity; and factual reliability through accuracy and citation coverage.",
          "task_data": "Public data availability is established. It contains 50 tasks, 25 structured or dynamic user profiles and 250 bilingual task–user queries; authors evaluate a 150-query subset.",
          "canonical_url": "https://arxiv.org/abs/2509.25106",
          "evaluation_code": "Public GitHub repository provides run and evaluation scripts and result directories, while the HF dataset is public; execution requires model and search API keys.",
          "judging_criteria": "GPT-5 judges personalization and quality; GPT-5-mini judges reliability. Criteria are dynamically generated, while reliability extracts claims and verifies retrieval and citation support.",
          "reported_results_provenance": "Comparisons cover commercial and open-source agents, search-augmented models and memory systems, but results are author-run; no independent benchmark results verified.",
          "reproducibility_and_barriers": "Queries, data and scripts are public, but proprietary judges, retrieval configuration, API access, dynamic context and live verification create practical reproducibility barriers.",
          "eligibility_date_and_evidence": "ArXiv v1 submitted September 29, 2025 (revised March 4, 2026), within cutoff; repository and HF dataset are public."
        },
        {
          "name": "Dr. Bench (formerly Rigorous Bench; Yao et al.)",
          "category": "Expert-curated long-form report benchmark",
          "evidence": [
            {
              "url": "https://openreview.net/forum?id=EYUG4Su6ZU",
              "supports": "Title, October 8, 2025 public record, ICLR submission and 214-query abstract."
            },
            {
              "url": "https://arxiv.org/abs/2510.02190",
              "supports": "Paper identity, date and canonical arXiv record."
            },
            {
              "url": "https://ar5iv.labs.arxiv.org/html/2510.02190",
              "supports": "Reference bundles, formulas, metrics and repository link."
            },
            {
              "url": "https://github.com/evigbyen/rigorousbench/",
              "supports": "Repository now uses Dr. Bench name; README does not establish turnkey evaluator availability."
            }
          ],
          "measures": "Measures semantic quality, topical drift, retrieval trustworthiness, contribution per token and retrieval index across 214 expert-curated queries in 10 domains.",
          "task_data": "Availability is partial: a public repository exists, but complete downloadable task and reference coverage was not verified. The paper describes manually constructed reference bundles.",
          "canonical_url": "https://arxiv.org/abs/2510.02190",
          "evaluation_code": "Public repository availability is established, but inspected materials provide an abstract and clone instructions only; a turnkey evaluator was not established, and the paper’s framework is not released scoring code.",
          "judging_criteria": "Semantic quality uses query-specific and general rubrics; topical focus penalizes missing or deviating anchor terms; retrieval trustworthiness checks exact and hostname matches against curated links.",
          "reported_results_provenance": "Thirteen-model comparisons and reported human agreement are authors' experiments; no independent reproduction verified.",
          "reproducibility_and_barriers": "Explicit formulas and curated references support offline scoring, but report parsing, link extraction and LLM rubric judgments introduce implementation and model sensitivity.",
          "eligibility_date_and_evidence": "Released October 2, 2025 through arXiv and repository; the October 2025 OpenReview submission used the earlier Rigorous Bench title."
        }
      ],
      "coverage_gaps": [
        "Coverage is broad but not proven exhaustive. Newly released, poorly indexed, non-English or privately distributed projects may be missing; the catalogue does not claim every qualifying project exists here.",
        "Artifact inspection establishes what was publicly documented or accessible, not that an end-to-end evaluation succeeds. “Not verified” means unresolved in the inspected sources, not proof that an artifact does not exist.",
        "Release dates and benchmark versions matter. Paper revisions, corrected answers, expanded datasets and changing repositories can yield different task counts; later vendor runs alone do not establish a new benchmark release.",
        "Complete-list scoring has different denominators: fixed expert gold, pooled discovered matches or a requested output quota. Recall and F1 therefore do not by themselves demonstrate exhaustive real-world coverage.",
        "Encrypted answers, locked test sets, missing corpora, unavailable proprietary components, paid API dependencies and model/judge drift can prevent exact independent reproduction even when some code is public.",
        "Public human preference or citation-quality evaluation is not automatically a test of autonomous discovery. Search Arena and SciArena are separated from agentic discovery benchmarks; broad science suites are included only for their relevant research components.",
        "Independence was not inferred from a third-party name, leaderboard entry or open repository. Commercial comparisons remain attributed to their authors; no cross-benchmark vendor ranking is warranted."
      ],
      "adjacent_or_uncertain": [
        {
          "url": "https://github.com/tavily-ai/tavily-search-evals",
          "name": "Tavily search-evals",
          "reason": "Public provider-comparison evaluation framework (SimpleQA/document relevance) released in the requested period, but not primarily comprehensive list-building or entity enrichment."
        },
        {
          "url": "https://doi.org/10.48550/arxiv.2506.01952",
          "name": "WebChoreArena",
          "reason": "June 2, 2025 benchmark with 532 human-curated tasks in simulated WebArena sites, emphasizing memory, calculation and tedious browser operations; functional task success, not web research or citation-backed synthesis."
        },
        {
          "url": "https://doi.org/10.48550/arxiv.2504.08942",
          "name": "AgentRewardBench",
          "reason": "April 11, 2025 benchmark of 1,302 web-agent trajectories for comparing automatic judges across five existing benchmarks; valuable evaluation-method artifact, but it evaluates judges/trajectories rather than research browsing itself."
        },
        {
          "url": "https://doi.org/10.48550/arxiv.2506.02865",
          "name": "WebVoyager updates / Surfer-H evaluation",
          "reason": "June 3, 2025 paper reports a 92.2% WebVoyager run and introduces WebVoyagerExtended (15,000 synthetic tasks/330 sites), but this is primarily an agent/model paper and browser-action benchmark update, not a new research-centric benchmark."
        },
        {
          "url": "https://doi.org/10.48550/arxiv.2510.02418",
          "name": "BrowserArena",
          "reason": "Adjacent browser-action evaluation. October 2, 2025 paper introduces live user-submitted web-navigation tasks, pairwise comparisons and step-level human feedback. The date is eligible, but navigation success and failure analysis—not research completeness or evidence-grounded enrichment—are its main target."
        },
        {
          "url": "https://github.com/perplexityai/search_evals",
          "name": "Perplexity search_evals",
          "reason": "Adjacent search-provider evaluation framework, released with the September 25, 2025 Search API report. Public runner, graders and traces reuse SimpleQA, FRAMES, BrowseComp and HLE. The associated provider comparisons are Perplexity/vendor runs, not new benchmarks or independent reproductions."
        },
        {
          "url": "https://github.com/parallel-web/parallel-llms-txt/blob/f6b31ffe/public/blog/deepsearch-qa.md",
          "name": "Parallel Task API DeepSearchQA evaluation",
          "reason": "2026 vendor report of running Google DeepSearchQA; not a new benchmark. Treat the reported results as Parallel’s runs, not independent reproduction."
        },
        {
          "url": "https://www.kaggle.com/benchmarks/google/facts-grounding",
          "name": "FACTS Grounding (v1; later Grounding v2)",
          "reason": "Static long-context grounded-answer evaluation rather than agentic web research; Grounding v2 should not be conflated with FACTS Search."
        },
        {
          "url": "https://github.com/OSU-NLP-Group/Online-Mind2Web",
          "name": "Online-Mind2Web",
          "reason": "Browser task/action completion on live websites, not primarily multi-source research, exhaustive discovery or evidence-supported enrichment."
        },
        {
          "url": "https://arxiv.org/abs/2409.12941",
          "name": "FRAMES (v3 / 2025 update)",
          "reason": "The original 824-question multi-hop Wikipedia benchmark predates 2025. A January 24, 2025 paper revision alone does not establish a substantive benchmark release. A 2026 community evaluator exists, but its provenance should not be conflated with an official new benchmark."
        },
        {
          "url": "https://arxiv.org/abs/2505.14558",
          "name": "R2MED",
          "reason": "Medical retrieval benchmark with public query/corpus/qrels and strong reasoning-centric design, but primarily a static closed-corpus retrieval benchmark rather than live web/literature research; excluded per scope."
        },
        {
          "url": "https://github.com/Talc-AI/search-bench",
          "name": "Talc-AI SearchBench",
          "reason": "Substantive public benchmark repository with 900 manually filtered Q&A items, four realistic categories and LLM-as-judge methodology, but its release/results are 2024 (scores as of 2024-08-30; launch 2024-09-19), outside the requested 2025-01-01–2026-09-26 eligibility window. It should not be counted as a qualifying record absent a qualifying 2025–26 update."
        },
        {
          "url": "https://github.com/Alibaba-NLP/DeepResearch/tree/main/WebAgent/WebWatcher",
          "name": "WebWatcher",
          "reason": "Agent/model project, not a second benchmark record. Its August 2025 paper introduces BrowseComp-VL, catalogued separately; other evaluation runs reuse existing datasets."
        },
        {
          "url": "https://github.com/google-deepmind/webquest",
          "name": "WebQuest",
          "reason": "Multimodal web-UI/page-sequence QA benchmark with public repository, but primary paper/repository are 2024 (outside eligibility window); static/browser-UI QA adjacent rather than core research/list-building."
        },
        {
          "url": "https://github.com/ServiceNow/drbench",
          "name": "DRBench (ServiceNow)",
          "reason": "Retained as an adjacent enterprise/internal-search benchmark: it does include public web sources, but evaluates cross-application private synthetic enterprise evidence as a defining requirement."
        },
        {
          "url": "https://github.com/aiming-lab/AutoResearchClaw/tree/main/experiments/arc_bench",
          "name": "ARC-Bench",
          "reason": "Scientific autonomous experimentation benchmark; research is explicit but web research is not the benchmark’s target."
        },
        {
          "url": "https://entityenricher.ai/docs/platform/benchmarks",
          "name": "Entity Enricher platform model benchmarks and benchmark scoring",
          "reason": "Vendor evaluation documentation for organization-specific saved entity/schema scenarios and model comparisons, not an established public dated benchmark release. Reference-based completeness/correctness scoring is described, but public tasks, results and evaluator artifacts were not verified."
        },
        {
          "url": "https://exa.ai/blog/websets-evals",
          "name": "Exa Websets benchmark",
          "reason": "Keep adjacent/vendor-only. The official February 19, 2025 post reports Exa's own comparison over 200 generated queries and GPT-4o grading, but no public raw dataset, evaluator or code was established; no independent validation should be implied."
        },
        {
          "url": "https://parallel.ai/products/findall",
          "name": "Parallel FindAll 40-query benchmark",
          "reason": "Vendor-only evaluation report: 40 discovery/enrichment queries, with recall measured against pooled correct matches from compared systems. Public task, gold and scorer artifacts and a firm release date were not established. Results are Parallel-created and reported, not independent."
        },
        {
          "url": "https://github.com/reka-ai/reka-vibe-eval",
          "name": "Reka Vibe-Eval",
          "reason": "Not eligible: multimodal chat benchmark repository was created in 2024 and evaluates image/multimodal generations, not web research; it is a false-positive despite the shared word 'Vibe'."
        },
        {
          "url": "https://www.vals.ai/benchmarks/web_search",
          "name": "Vals Web Search Index",
          "reason": "July 16, 2026 controlled search-tool comparison on legal research and finance tasks. It reports 208 legal tasks and 450 finance questions, with rubric-based grading. Keep as a public comparative report: a standalone reusable task/evaluator package and organizational independence were not established; linked orchestration repositories alone do not establish open task data."
        },
        {
          "url": "https://huggingface.co/datasets/jxg25/EntiWeave",
          "name": "EntiWeave",
          "reason": "Public graph-grounded web-search training data and a 100-question held-out evaluation split were found, but a qualifying release/update date and benchmark evaluator were not established. Retained as uncertain, not counted as dated core coverage."
        },
        {
          "url": "https://aclanthology.org/2025.findings-acl.988/",
          "name": "X-WebAgentBench: A Multilingual Interactive Web Benchmark for Evaluating Global Agentic System",
          "reason": "ACL Findings 2025 multilingual interactive-web benchmark. Evaluates instruction following and product/website interaction using WebShop-style task success, not research discovery or citation-backed synthesis. Public paper verified; data/evaluator release not established here."
        },
        {
          "url": "https://arxiv.org/abs/2506.05334",
          "name": "Search Arena",
          "reason": "2025 public search-LLM preference platform with conversation/vote data and analysis code. Kept adjacent because its primary outcome is pairwise user preference on search chats, rather than objective discovery correctness, evidence coverage or set completeness. Votes were collected by the authors; they are not independent reproduction."
        },
        {
          "url": "https://arxiv.org/abs/2604.14683",
          "name": "DR³-Eval",
          "reason": "April 2026 public code/data benchmark for multimodal, multi-file research reports in static task sandboxes. Scores information recall, factual accuracy, citations, instruction following and depth. Adjacent because supplied files/controlled workspaces, rather than open-web discovery, define its tasks."
        },
        {
          "url": "https://arxiv.org/abs/2602.15019",
          "name": "Hunt Globally: Wide Search AI Agents for Drug Asset Scouting in Investing, Business Development, and Competitive Intelligence",
          "reason": "Uncertain standalone benchmark: February 16, 2026 drug-asset scouting paper describes 48 seed queries and 22 held-out query–asset pairs, with asset precision/recall/F1 and expert-calibrated LLM grading. Results are author-run; public task, gold and evaluator releases were not verified. Retained as a paper-defined evaluation study, not a confirmed reusable public benchmark."
        },
        {
          "url": "https://ai.meta.com/research/publications/gaia-a-benchmark-for-general-ai-assistants/",
          "name": "GAIA",
          "reason": "Relevant general-assistant benchmark with web browsing, but the original paper/release is 2023–2024. Later agent runs or leaderboard submissions do not themselves demonstrate a substantive 2025–September 2026 benchmark update; none was established here."
        },
        {
          "url": "https://arxiv.org/abs/2501.14249",
          "name": "Humanity’s Last Exam (HLE)",
          "reason": "Eligible 2025 release, but primarily closed-ended expert academic questions testing broad knowledge/reasoning, not open-web research, complete entity discovery or evidence-supported enrichment. Agent papers using search on HLE are benchmark runs, not new research benchmarks."
        },
        {
          "url": "https://arxiv.org/html/2509.07968v1",
          "name": "SimpleQA Verified",
          "reason": "September 9, 2025 factuality benchmark designed to measure parametric knowledge without tools. Kept separate from agentic web research despite use of related SimpleQA datasets in search-provider evaluations."
        },
        {
          "url": "https://github.com/yale-nlp/SciArena",
          "name": "SciArena and SciArena-Eval",
          "reason": "Public July 1, 2025 release of platform, preference data and analysis code; later paper includes a meta-evaluation benchmark. Adjacent: primarily compares foundation-model literature-grounded responses and judge agreement with human votes, rather than autonomous discovery or complete-set retrieval. Citation-attribution analysis is relevant, and paper-bank data are available, but preferences are not evidence-completeness scores."
        }
      ]
    },
    "grounding": [
      {
        "field": "structured.benchmarks[0].name",
        "citations": [
          {
            "url": "https://openai.com/index/browsecomp/"
          },
          {
            "url": "https://github.com/openai/simple-evals/blob/main/browsecomp_eval.py"
          },
          {
            "url": "https://arxiv.org/abs/2504.12516"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[0].canonical_url",
        "citations": [
          {
            "url": "https://openai.com/index/browsecomp/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[0].category",
        "citations": [
          {
            "url": "https://openai.com/index/browsecomp/"
          },
          {
            "url": "https://github.com/openai/simple-evals/blob/main/browsecomp_eval.py"
          },
          {
            "url": "https://arxiv.org/abs/2504.12516"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[0].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://openai.com/index/browsecomp/"
          },
          {
            "url": "https://github.com/openai/simple-evals/blob/main/browsecomp_eval.py"
          },
          {
            "url": "https://arxiv.org/abs/2504.12516"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[0].measures",
        "citations": [
          {
            "url": "https://openai.com/index/browsecomp/"
          },
          {
            "url": "https://github.com/openai/simple-evals/blob/main/browsecomp_eval.py"
          },
          {
            "url": "https://arxiv.org/abs/2504.12516"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[0].task_data",
        "citations": [
          {
            "url": "https://openai.com/index/browsecomp/"
          },
          {
            "url": "https://github.com/openai/simple-evals/blob/main/browsecomp_eval.py"
          },
          {
            "url": "https://arxiv.org/abs/2504.12516"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[0].evaluation_code",
        "citations": [
          {
            "url": "https://openai.com/index/browsecomp/"
          },
          {
            "url": "https://github.com/openai/simple-evals/blob/main/browsecomp_eval.py"
          },
          {
            "url": "https://arxiv.org/abs/2504.12516"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[0].judging_criteria",
        "citations": [
          {
            "url": "https://openai.com/index/browsecomp/"
          },
          {
            "url": "https://github.com/openai/simple-evals/blob/main/browsecomp_eval.py"
          },
          {
            "url": "https://arxiv.org/abs/2504.12516"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[0].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://openai.com/index/browsecomp/"
          },
          {
            "url": "https://github.com/openai/simple-evals/blob/main/browsecomp_eval.py"
          },
          {
            "url": "https://arxiv.org/abs/2504.12516"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[0].reported_results_provenance",
        "citations": [
          {
            "url": "https://openai.com/index/browsecomp/"
          },
          {
            "url": "https://github.com/openai/simple-evals/blob/main/browsecomp_eval.py"
          },
          {
            "url": "https://arxiv.org/abs/2504.12516"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[0].evidence",
        "citations": [
          {
            "url": "https://openai.com/index/browsecomp/"
          },
          {
            "url": "https://github.com/openai/simple-evals/blob/main/browsecomp_eval.py"
          },
          {
            "url": "https://arxiv.org/abs/2504.12516"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[1].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://arxiv.org/html/2508.07999"
          },
          {
            "url": "https://github.com/ByteDance-Seed/WideSearch"
          },
          {
            "url": "https://widesearch-seed.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[1].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[1].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://arxiv.org/html/2508.07999"
          },
          {
            "url": "https://github.com/ByteDance-Seed/WideSearch"
          },
          {
            "url": "https://widesearch-seed.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[1].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://arxiv.org/html/2508.07999"
          },
          {
            "url": "https://github.com/ByteDance-Seed/WideSearch"
          },
          {
            "url": "https://widesearch-seed.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[1].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://arxiv.org/html/2508.07999"
          },
          {
            "url": "https://github.com/ByteDance-Seed/WideSearch"
          },
          {
            "url": "https://widesearch-seed.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[1].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://arxiv.org/html/2508.07999"
          },
          {
            "url": "https://github.com/ByteDance-Seed/WideSearch"
          },
          {
            "url": "https://widesearch-seed.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[1].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://arxiv.org/html/2508.07999"
          },
          {
            "url": "https://github.com/ByteDance-Seed/WideSearch"
          },
          {
            "url": "https://widesearch-seed.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[1].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://arxiv.org/html/2508.07999"
          },
          {
            "url": "https://github.com/ByteDance-Seed/WideSearch"
          },
          {
            "url": "https://widesearch-seed.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[1].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://arxiv.org/html/2508.07999"
          },
          {
            "url": "https://github.com/ByteDance-Seed/WideSearch"
          },
          {
            "url": "https://widesearch-seed.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[1].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://arxiv.org/html/2508.07999"
          },
          {
            "url": "https://github.com/ByteDance-Seed/WideSearch"
          },
          {
            "url": "https://widesearch-seed.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[1].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://arxiv.org/html/2508.07999"
          },
          {
            "url": "https://github.com/ByteDance-Seed/WideSearch"
          },
          {
            "url": "https://widesearch-seed.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/ByteDance-Seed/WideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[2].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.27595"
          },
          {
            "url": "https://arxiv.org/html/2606.27595"
          },
          {
            "url": "https://github.com/minstar/Ko-widesearch"
          },
          {
            "url": "https://minstar.github.io/Ko-widesearch/"
          },
          {
            "url": "https://huggingface.co/datasets/Minbyul/Ko-widesearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[2].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.27595"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[2].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.27595"
          },
          {
            "url": "https://arxiv.org/html/2606.27595"
          },
          {
            "url": "https://github.com/minstar/Ko-widesearch"
          },
          {
            "url": "https://minstar.github.io/Ko-widesearch/"
          },
          {
            "url": "https://huggingface.co/datasets/Minbyul/Ko-widesearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[2].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.27595"
          },
          {
            "url": "https://arxiv.org/html/2606.27595"
          },
          {
            "url": "https://github.com/minstar/Ko-widesearch"
          },
          {
            "url": "https://minstar.github.io/Ko-widesearch/"
          },
          {
            "url": "https://huggingface.co/datasets/Minbyul/Ko-widesearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[2].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.27595"
          },
          {
            "url": "https://arxiv.org/html/2606.27595"
          },
          {
            "url": "https://github.com/minstar/Ko-widesearch"
          },
          {
            "url": "https://minstar.github.io/Ko-widesearch/"
          },
          {
            "url": "https://huggingface.co/datasets/Minbyul/Ko-widesearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[2].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.27595"
          },
          {
            "url": "https://arxiv.org/html/2606.27595"
          },
          {
            "url": "https://github.com/minstar/Ko-widesearch"
          },
          {
            "url": "https://minstar.github.io/Ko-widesearch/"
          },
          {
            "url": "https://huggingface.co/datasets/Minbyul/Ko-widesearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[2].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.27595"
          },
          {
            "url": "https://arxiv.org/html/2606.27595"
          },
          {
            "url": "https://github.com/minstar/Ko-widesearch"
          },
          {
            "url": "https://minstar.github.io/Ko-widesearch/"
          },
          {
            "url": "https://huggingface.co/datasets/Minbyul/Ko-widesearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[2].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.27595"
          },
          {
            "url": "https://arxiv.org/html/2606.27595"
          },
          {
            "url": "https://github.com/minstar/Ko-widesearch"
          },
          {
            "url": "https://minstar.github.io/Ko-widesearch/"
          },
          {
            "url": "https://huggingface.co/datasets/Minbyul/Ko-widesearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[2].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.27595"
          },
          {
            "url": "https://arxiv.org/html/2606.27595"
          },
          {
            "url": "https://github.com/minstar/Ko-widesearch"
          },
          {
            "url": "https://minstar.github.io/Ko-widesearch/"
          },
          {
            "url": "https://huggingface.co/datasets/Minbyul/Ko-widesearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[2].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.27595"
          },
          {
            "url": "https://arxiv.org/html/2606.27595"
          },
          {
            "url": "https://github.com/minstar/Ko-widesearch"
          },
          {
            "url": "https://minstar.github.io/Ko-widesearch/"
          },
          {
            "url": "https://huggingface.co/datasets/Minbyul/Ko-widesearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[2].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.27595"
          },
          {
            "url": "https://arxiv.org/html/2606.27595"
          },
          {
            "url": "https://github.com/minstar/Ko-widesearch"
          },
          {
            "url": "https://minstar.github.io/Ko-widesearch/"
          },
          {
            "url": "https://huggingface.co/datasets/Minbyul/Ko-widesearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[3].name",
        "citations": [
          {
            "url": "https://arxiv.org/html/2504.12682"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2504.12682"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[3].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/html/2504.12682"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[3].category",
        "citations": [
          {
            "url": "https://arxiv.org/html/2504.12682"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2504.12682"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[3].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/html/2504.12682"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2504.12682"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[3].measures",
        "citations": [
          {
            "url": "https://arxiv.org/html/2504.12682"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2504.12682"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[3].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/html/2504.12682"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2504.12682"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[3].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/html/2504.12682"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2504.12682"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[3].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/html/2504.12682"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2504.12682"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[3].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/html/2504.12682"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2504.12682"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[3].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/html/2504.12682"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2504.12682"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[3].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/html/2504.12682"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2504.12682"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[4].name",
        "citations": [
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://arxiv.org/abs/2506.11763"
          },
          {
            "url": "https://deepresearch-bench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[4].canonical_url",
        "citations": [
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[4].category",
        "citations": [
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://arxiv.org/abs/2506.11763"
          },
          {
            "url": "https://deepresearch-bench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[4].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://arxiv.org/abs/2506.11763"
          },
          {
            "url": "https://deepresearch-bench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[4].measures",
        "citations": [
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://arxiv.org/abs/2506.11763"
          },
          {
            "url": "https://deepresearch-bench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[4].task_data",
        "citations": [
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://arxiv.org/abs/2506.11763"
          },
          {
            "url": "https://deepresearch-bench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[4].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://arxiv.org/abs/2506.11763"
          },
          {
            "url": "https://deepresearch-bench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[4].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://arxiv.org/abs/2506.11763"
          },
          {
            "url": "https://deepresearch-bench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[4].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://arxiv.org/abs/2506.11763"
          },
          {
            "url": "https://deepresearch-bench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[4].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://arxiv.org/abs/2506.11763"
          },
          {
            "url": "https://deepresearch-bench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[4].evidence",
        "citations": [
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://arxiv.org/abs/2506.11763"
          },
          {
            "url": "https://deepresearch-bench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[5].name",
        "citations": [
          {
            "url": "https://github.com/imlrz/DeepResearch-Bench-II"
          },
          {
            "url": "https://arxiv.org/abs/2601.08536"
          },
          {
            "url": "https://arxiv.org/html/2601.08536"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[5].canonical_url",
        "citations": [
          {
            "url": "https://github.com/imlrz/DeepResearch-Bench-II"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[5].category",
        "citations": [
          {
            "url": "https://github.com/imlrz/DeepResearch-Bench-II"
          },
          {
            "url": "https://arxiv.org/abs/2601.08536"
          },
          {
            "url": "https://arxiv.org/html/2601.08536"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[5].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/imlrz/DeepResearch-Bench-II"
          },
          {
            "url": "https://arxiv.org/abs/2601.08536"
          },
          {
            "url": "https://arxiv.org/html/2601.08536"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[5].measures",
        "citations": [
          {
            "url": "https://github.com/imlrz/DeepResearch-Bench-II"
          },
          {
            "url": "https://arxiv.org/abs/2601.08536"
          },
          {
            "url": "https://arxiv.org/html/2601.08536"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[5].task_data",
        "citations": [
          {
            "url": "https://github.com/imlrz/DeepResearch-Bench-II"
          },
          {
            "url": "https://arxiv.org/abs/2601.08536"
          },
          {
            "url": "https://arxiv.org/html/2601.08536"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[5].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/imlrz/DeepResearch-Bench-II"
          },
          {
            "url": "https://arxiv.org/abs/2601.08536"
          },
          {
            "url": "https://arxiv.org/html/2601.08536"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[5].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/imlrz/DeepResearch-Bench-II"
          },
          {
            "url": "https://arxiv.org/abs/2601.08536"
          },
          {
            "url": "https://arxiv.org/html/2601.08536"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[5].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/imlrz/DeepResearch-Bench-II"
          },
          {
            "url": "https://arxiv.org/abs/2601.08536"
          },
          {
            "url": "https://arxiv.org/html/2601.08536"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[5].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/imlrz/DeepResearch-Bench-II"
          },
          {
            "url": "https://arxiv.org/abs/2601.08536"
          },
          {
            "url": "https://arxiv.org/html/2601.08536"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[5].evidence",
        "citations": [
          {
            "url": "https://github.com/imlrz/DeepResearch-Bench-II"
          },
          {
            "url": "https://arxiv.org/abs/2601.08536"
          },
          {
            "url": "https://arxiv.org/html/2601.08536"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[6].name",
        "citations": [
          {
            "url": "https://github.com/HKUDS/DeepResearch-Eval"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2510.07861"
          },
          {
            "url": "https://arxiv.org/pdf/2510.07861"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[6].canonical_url",
        "citations": [
          {
            "url": "https://github.com/HKUDS/DeepResearch-Eval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[6].category",
        "citations": [
          {
            "url": "https://github.com/HKUDS/DeepResearch-Eval"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2510.07861"
          },
          {
            "url": "https://arxiv.org/pdf/2510.07861"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[6].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/HKUDS/DeepResearch-Eval"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2510.07861"
          },
          {
            "url": "https://arxiv.org/pdf/2510.07861"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[6].measures",
        "citations": [
          {
            "url": "https://github.com/HKUDS/DeepResearch-Eval"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2510.07861"
          },
          {
            "url": "https://arxiv.org/pdf/2510.07861"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[6].task_data",
        "citations": [
          {
            "url": "https://github.com/HKUDS/DeepResearch-Eval"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2510.07861"
          },
          {
            "url": "https://arxiv.org/pdf/2510.07861"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[6].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/HKUDS/DeepResearch-Eval"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2510.07861"
          },
          {
            "url": "https://arxiv.org/pdf/2510.07861"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[6].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/HKUDS/DeepResearch-Eval"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2510.07861"
          },
          {
            "url": "https://arxiv.org/pdf/2510.07861"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[6].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/HKUDS/DeepResearch-Eval"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2510.07861"
          },
          {
            "url": "https://arxiv.org/pdf/2510.07861"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[6].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/HKUDS/DeepResearch-Eval"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2510.07861"
          },
          {
            "url": "https://arxiv.org/pdf/2510.07861"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[6].evidence",
        "citations": [
          {
            "url": "https://github.com/HKUDS/DeepResearch-Eval"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2510.07861"
          },
          {
            "url": "https://arxiv.org/pdf/2510.07861"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[7].name",
        "citations": [
          {
            "url": "https://github.com/GAIR-NLP/ResearcherBench"
          },
          {
            "url": "https://arxiv.org/abs/2507.16280"
          },
          {
            "url": "https://researcherbench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[7].canonical_url",
        "citations": [
          {
            "url": "https://github.com/GAIR-NLP/ResearcherBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[7].category",
        "citations": [
          {
            "url": "https://github.com/GAIR-NLP/ResearcherBench"
          },
          {
            "url": "https://arxiv.org/abs/2507.16280"
          },
          {
            "url": "https://researcherbench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[7].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/GAIR-NLP/ResearcherBench"
          },
          {
            "url": "https://arxiv.org/abs/2507.16280"
          },
          {
            "url": "https://researcherbench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[7].measures",
        "citations": [
          {
            "url": "https://github.com/GAIR-NLP/ResearcherBench"
          },
          {
            "url": "https://arxiv.org/abs/2507.16280"
          },
          {
            "url": "https://researcherbench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[7].task_data",
        "citations": [
          {
            "url": "https://github.com/GAIR-NLP/ResearcherBench"
          },
          {
            "url": "https://arxiv.org/abs/2507.16280"
          },
          {
            "url": "https://researcherbench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[7].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/GAIR-NLP/ResearcherBench"
          },
          {
            "url": "https://arxiv.org/abs/2507.16280"
          },
          {
            "url": "https://researcherbench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[7].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/GAIR-NLP/ResearcherBench"
          },
          {
            "url": "https://arxiv.org/abs/2507.16280"
          },
          {
            "url": "https://researcherbench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[7].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/GAIR-NLP/ResearcherBench"
          },
          {
            "url": "https://arxiv.org/abs/2507.16280"
          },
          {
            "url": "https://researcherbench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[7].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/GAIR-NLP/ResearcherBench"
          },
          {
            "url": "https://arxiv.org/abs/2507.16280"
          },
          {
            "url": "https://researcherbench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[7].evidence",
        "citations": [
          {
            "url": "https://github.com/GAIR-NLP/ResearcherBench"
          },
          {
            "url": "https://arxiv.org/abs/2507.16280"
          },
          {
            "url": "https://researcherbench.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[8].name",
        "citations": [
          {
            "url": "https://github.com/scaleapi/researchrubrics"
          },
          {
            "url": "https://arxiv.org/html/2511.07685"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[8].canonical_url",
        "citations": [
          {
            "url": "https://github.com/scaleapi/researchrubrics"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[8].category",
        "citations": [
          {
            "url": "https://github.com/scaleapi/researchrubrics"
          },
          {
            "url": "https://arxiv.org/html/2511.07685"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[8].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/scaleapi/researchrubrics"
          },
          {
            "url": "https://arxiv.org/html/2511.07685"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[8].measures",
        "citations": [
          {
            "url": "https://github.com/scaleapi/researchrubrics"
          },
          {
            "url": "https://arxiv.org/html/2511.07685"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[8].task_data",
        "citations": [
          {
            "url": "https://github.com/scaleapi/researchrubrics"
          },
          {
            "url": "https://arxiv.org/html/2511.07685"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[8].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/scaleapi/researchrubrics"
          },
          {
            "url": "https://arxiv.org/html/2511.07685"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[8].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/scaleapi/researchrubrics"
          },
          {
            "url": "https://arxiv.org/html/2511.07685"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[8].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/scaleapi/researchrubrics"
          },
          {
            "url": "https://arxiv.org/html/2511.07685"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[8].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/scaleapi/researchrubrics"
          },
          {
            "url": "https://arxiv.org/html/2511.07685"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[8].evidence",
        "citations": [
          {
            "url": "https://github.com/scaleapi/researchrubrics"
          },
          {
            "url": "https://arxiv.org/html/2511.07685"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[9].name",
        "citations": [
          {
            "url": "https://github.com/SalesforceAIResearch/LiveResearchBench"
          },
          {
            "url": "https://arxiv.org/abs/2510.14240"
          },
          {
            "url": "https://huggingface.co/datasets/Salesforce/LiveResearchBench"
          },
          {
            "url": "https://livedeepresearch.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[9].canonical_url",
        "citations": [
          {
            "url": "https://github.com/SalesforceAIResearch/LiveResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[9].category",
        "citations": [
          {
            "url": "https://github.com/SalesforceAIResearch/LiveResearchBench"
          },
          {
            "url": "https://arxiv.org/abs/2510.14240"
          },
          {
            "url": "https://huggingface.co/datasets/Salesforce/LiveResearchBench"
          },
          {
            "url": "https://livedeepresearch.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[9].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/SalesforceAIResearch/LiveResearchBench"
          },
          {
            "url": "https://arxiv.org/abs/2510.14240"
          },
          {
            "url": "https://huggingface.co/datasets/Salesforce/LiveResearchBench"
          },
          {
            "url": "https://livedeepresearch.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[9].measures",
        "citations": [
          {
            "url": "https://github.com/SalesforceAIResearch/LiveResearchBench"
          },
          {
            "url": "https://arxiv.org/abs/2510.14240"
          },
          {
            "url": "https://huggingface.co/datasets/Salesforce/LiveResearchBench"
          },
          {
            "url": "https://livedeepresearch.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[9].task_data",
        "citations": [
          {
            "url": "https://github.com/SalesforceAIResearch/LiveResearchBench"
          },
          {
            "url": "https://arxiv.org/abs/2510.14240"
          },
          {
            "url": "https://huggingface.co/datasets/Salesforce/LiveResearchBench"
          },
          {
            "url": "https://livedeepresearch.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[9].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/SalesforceAIResearch/LiveResearchBench"
          },
          {
            "url": "https://arxiv.org/abs/2510.14240"
          },
          {
            "url": "https://huggingface.co/datasets/Salesforce/LiveResearchBench"
          },
          {
            "url": "https://livedeepresearch.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[9].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/SalesforceAIResearch/LiveResearchBench"
          },
          {
            "url": "https://arxiv.org/abs/2510.14240"
          },
          {
            "url": "https://huggingface.co/datasets/Salesforce/LiveResearchBench"
          },
          {
            "url": "https://livedeepresearch.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[9].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/SalesforceAIResearch/LiveResearchBench"
          },
          {
            "url": "https://arxiv.org/abs/2510.14240"
          },
          {
            "url": "https://huggingface.co/datasets/Salesforce/LiveResearchBench"
          },
          {
            "url": "https://livedeepresearch.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[9].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/SalesforceAIResearch/LiveResearchBench"
          },
          {
            "url": "https://arxiv.org/abs/2510.14240"
          },
          {
            "url": "https://huggingface.co/datasets/Salesforce/LiveResearchBench"
          },
          {
            "url": "https://livedeepresearch.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[9].evidence",
        "citations": [
          {
            "url": "https://github.com/SalesforceAIResearch/LiveResearchBench"
          },
          {
            "url": "https://arxiv.org/abs/2510.14240"
          },
          {
            "url": "https://huggingface.co/datasets/Salesforce/LiveResearchBench"
          },
          {
            "url": "https://livedeepresearch.github.io/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[10].name",
        "citations": [
          {
            "url": "https://github.com/ByteDance-BandAI/ReportBench"
          },
          {
            "url": "https://openreview.net/forum?id=zvL42fmtbG"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[10].canonical_url",
        "citations": [
          {
            "url": "https://github.com/ByteDance-BandAI/ReportBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[10].category",
        "citations": [
          {
            "url": "https://github.com/ByteDance-BandAI/ReportBench"
          },
          {
            "url": "https://openreview.net/forum?id=zvL42fmtbG"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[10].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/ByteDance-BandAI/ReportBench"
          },
          {
            "url": "https://openreview.net/forum?id=zvL42fmtbG"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[10].measures",
        "citations": [
          {
            "url": "https://github.com/ByteDance-BandAI/ReportBench"
          },
          {
            "url": "https://openreview.net/forum?id=zvL42fmtbG"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[10].task_data",
        "citations": [
          {
            "url": "https://github.com/ByteDance-BandAI/ReportBench"
          },
          {
            "url": "https://openreview.net/forum?id=zvL42fmtbG"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[10].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/ByteDance-BandAI/ReportBench"
          },
          {
            "url": "https://openreview.net/forum?id=zvL42fmtbG"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[10].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/ByteDance-BandAI/ReportBench"
          },
          {
            "url": "https://openreview.net/forum?id=zvL42fmtbG"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[10].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/ByteDance-BandAI/ReportBench"
          },
          {
            "url": "https://openreview.net/forum?id=zvL42fmtbG"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[10].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/ByteDance-BandAI/ReportBench"
          },
          {
            "url": "https://openreview.net/forum?id=zvL42fmtbG"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[10].evidence",
        "citations": [
          {
            "url": "https://github.com/ByteDance-BandAI/ReportBench"
          },
          {
            "url": "https://openreview.net/forum?id=zvL42fmtbG"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[11].name",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard"
          },
          {
            "url": "https://deepmind.google/blog/facts-benchmark-suite-systematically-evaluating-the-factuality-of-large-language-models/"
          },
          {
            "url": "https://arxiv.org/abs/2512.10791"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[11].canonical_url",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[11].category",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard"
          },
          {
            "url": "https://deepmind.google/blog/facts-benchmark-suite-systematically-evaluating-the-factuality-of-large-language-models/"
          },
          {
            "url": "https://arxiv.org/abs/2512.10791"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[11].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard"
          },
          {
            "url": "https://deepmind.google/blog/facts-benchmark-suite-systematically-evaluating-the-factuality-of-large-language-models/"
          },
          {
            "url": "https://arxiv.org/abs/2512.10791"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[11].measures",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard"
          },
          {
            "url": "https://deepmind.google/blog/facts-benchmark-suite-systematically-evaluating-the-factuality-of-large-language-models/"
          },
          {
            "url": "https://arxiv.org/abs/2512.10791"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[11].task_data",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard"
          },
          {
            "url": "https://deepmind.google/blog/facts-benchmark-suite-systematically-evaluating-the-factuality-of-large-language-models/"
          },
          {
            "url": "https://arxiv.org/abs/2512.10791"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[11].evaluation_code",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard"
          },
          {
            "url": "https://deepmind.google/blog/facts-benchmark-suite-systematically-evaluating-the-factuality-of-large-language-models/"
          },
          {
            "url": "https://arxiv.org/abs/2512.10791"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[11].judging_criteria",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard"
          },
          {
            "url": "https://deepmind.google/blog/facts-benchmark-suite-systematically-evaluating-the-factuality-of-large-language-models/"
          },
          {
            "url": "https://arxiv.org/abs/2512.10791"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[11].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard"
          },
          {
            "url": "https://deepmind.google/blog/facts-benchmark-suite-systematically-evaluating-the-factuality-of-large-language-models/"
          },
          {
            "url": "https://arxiv.org/abs/2512.10791"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[11].reported_results_provenance",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard"
          },
          {
            "url": "https://deepmind.google/blog/facts-benchmark-suite-systematically-evaluating-the-factuality-of-large-language-models/"
          },
          {
            "url": "https://arxiv.org/abs/2512.10791"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[11].evidence",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-search/leaderboard"
          },
          {
            "url": "https://deepmind.google/blog/facts-benchmark-suite-systematically-evaluating-the-factuality-of-large-language-models/"
          },
          {
            "url": "https://arxiv.org/abs/2512.10791"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[12].name",
        "citations": [
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://arxiv.org/html/2508.06600v1"
          },
          {
            "url": "https://texttron.github.io/BrowseComp-Plus/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[12].canonical_url",
        "citations": [
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[12].category",
        "citations": [
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://arxiv.org/html/2508.06600v1"
          },
          {
            "url": "https://texttron.github.io/BrowseComp-Plus/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[12].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://arxiv.org/html/2508.06600v1"
          },
          {
            "url": "https://texttron.github.io/BrowseComp-Plus/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[12].measures",
        "citations": [
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://arxiv.org/html/2508.06600v1"
          },
          {
            "url": "https://texttron.github.io/BrowseComp-Plus/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[12].task_data",
        "citations": [
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://arxiv.org/html/2508.06600v1"
          },
          {
            "url": "https://texttron.github.io/BrowseComp-Plus/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[12].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://arxiv.org/html/2508.06600v1"
          },
          {
            "url": "https://texttron.github.io/BrowseComp-Plus/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[12].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://arxiv.org/html/2508.06600v1"
          },
          {
            "url": "https://texttron.github.io/BrowseComp-Plus/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[12].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://arxiv.org/html/2508.06600v1"
          },
          {
            "url": "https://texttron.github.io/BrowseComp-Plus/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[12].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://arxiv.org/html/2508.06600v1"
          },
          {
            "url": "https://texttron.github.io/BrowseComp-Plus/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[12].evidence",
        "citations": [
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://arxiv.org/html/2508.06600v1"
          },
          {
            "url": "https://texttron.github.io/BrowseComp-Plus/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[13].name",
        "citations": [
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          },
          {
            "url": "https://arxiv.org/html/2504.19314"
          },
          {
            "url": "https://github.com/open-compass/AgentCompass/blob/a7c30989/src/agentcompass/benchmarks/browsecomp_zh.py"
          },
          {
            "url": "https://github.com/AGI-Eval-Official/BrowseComp-ZH-revised"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[13].canonical_url",
        "citations": [
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[13].category",
        "citations": [
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          },
          {
            "url": "https://arxiv.org/html/2504.19314"
          },
          {
            "url": "https://github.com/open-compass/AgentCompass/blob/a7c30989/src/agentcompass/benchmarks/browsecomp_zh.py"
          },
          {
            "url": "https://github.com/AGI-Eval-Official/BrowseComp-ZH-revised"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[13].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          },
          {
            "url": "https://arxiv.org/html/2504.19314"
          },
          {
            "url": "https://github.com/open-compass/AgentCompass/blob/a7c30989/src/agentcompass/benchmarks/browsecomp_zh.py"
          },
          {
            "url": "https://github.com/AGI-Eval-Official/BrowseComp-ZH-revised"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[13].measures",
        "citations": [
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          },
          {
            "url": "https://arxiv.org/html/2504.19314"
          },
          {
            "url": "https://github.com/open-compass/AgentCompass/blob/a7c30989/src/agentcompass/benchmarks/browsecomp_zh.py"
          },
          {
            "url": "https://github.com/AGI-Eval-Official/BrowseComp-ZH-revised"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[13].task_data",
        "citations": [
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          },
          {
            "url": "https://arxiv.org/html/2504.19314"
          },
          {
            "url": "https://github.com/open-compass/AgentCompass/blob/a7c30989/src/agentcompass/benchmarks/browsecomp_zh.py"
          },
          {
            "url": "https://github.com/AGI-Eval-Official/BrowseComp-ZH-revised"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[13].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          },
          {
            "url": "https://arxiv.org/html/2504.19314"
          },
          {
            "url": "https://github.com/open-compass/AgentCompass/blob/a7c30989/src/agentcompass/benchmarks/browsecomp_zh.py"
          },
          {
            "url": "https://github.com/AGI-Eval-Official/BrowseComp-ZH-revised"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[13].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          },
          {
            "url": "https://arxiv.org/html/2504.19314"
          },
          {
            "url": "https://github.com/open-compass/AgentCompass/blob/a7c30989/src/agentcompass/benchmarks/browsecomp_zh.py"
          },
          {
            "url": "https://github.com/AGI-Eval-Official/BrowseComp-ZH-revised"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[13].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          },
          {
            "url": "https://arxiv.org/html/2504.19314"
          },
          {
            "url": "https://github.com/open-compass/AgentCompass/blob/a7c30989/src/agentcompass/benchmarks/browsecomp_zh.py"
          },
          {
            "url": "https://github.com/AGI-Eval-Official/BrowseComp-ZH-revised"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[13].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          },
          {
            "url": "https://arxiv.org/html/2504.19314"
          },
          {
            "url": "https://github.com/open-compass/AgentCompass/blob/a7c30989/src/agentcompass/benchmarks/browsecomp_zh.py"
          },
          {
            "url": "https://github.com/AGI-Eval-Official/BrowseComp-ZH-revised"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[13].evidence",
        "citations": [
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          },
          {
            "url": "https://arxiv.org/html/2504.19314"
          },
          {
            "url": "https://github.com/open-compass/AgentCompass/blob/a7c30989/src/agentcompass/benchmarks/browsecomp_zh.py"
          },
          {
            "url": "https://github.com/AGI-Eval-Official/BrowseComp-ZH-revised"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[14].name",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/WebAgent"
          },
          {
            "url": "https://arxiv.org/abs/2501.07572"
          },
          {
            "url": "https://aclanthology.org/2025.acl-long.508.pdf"
          },
          {
            "url": "https://huggingface.co/datasets/callanwu/WebWalkerQA"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[14].canonical_url",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/WebAgent"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[14].category",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/WebAgent"
          },
          {
            "url": "https://arxiv.org/abs/2501.07572"
          },
          {
            "url": "https://aclanthology.org/2025.acl-long.508.pdf"
          },
          {
            "url": "https://huggingface.co/datasets/callanwu/WebWalkerQA"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[14].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/WebAgent"
          },
          {
            "url": "https://arxiv.org/abs/2501.07572"
          },
          {
            "url": "https://aclanthology.org/2025.acl-long.508.pdf"
          },
          {
            "url": "https://huggingface.co/datasets/callanwu/WebWalkerQA"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[14].measures",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/WebAgent"
          },
          {
            "url": "https://arxiv.org/abs/2501.07572"
          },
          {
            "url": "https://aclanthology.org/2025.acl-long.508.pdf"
          },
          {
            "url": "https://huggingface.co/datasets/callanwu/WebWalkerQA"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[14].task_data",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/WebAgent"
          },
          {
            "url": "https://arxiv.org/abs/2501.07572"
          },
          {
            "url": "https://aclanthology.org/2025.acl-long.508.pdf"
          },
          {
            "url": "https://huggingface.co/datasets/callanwu/WebWalkerQA"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[14].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/WebAgent"
          },
          {
            "url": "https://arxiv.org/abs/2501.07572"
          },
          {
            "url": "https://aclanthology.org/2025.acl-long.508.pdf"
          },
          {
            "url": "https://huggingface.co/datasets/callanwu/WebWalkerQA"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[14].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/WebAgent"
          },
          {
            "url": "https://arxiv.org/abs/2501.07572"
          },
          {
            "url": "https://aclanthology.org/2025.acl-long.508.pdf"
          },
          {
            "url": "https://huggingface.co/datasets/callanwu/WebWalkerQA"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[14].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/WebAgent"
          },
          {
            "url": "https://arxiv.org/abs/2501.07572"
          },
          {
            "url": "https://aclanthology.org/2025.acl-long.508.pdf"
          },
          {
            "url": "https://huggingface.co/datasets/callanwu/WebWalkerQA"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[14].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/WebAgent"
          },
          {
            "url": "https://arxiv.org/abs/2501.07572"
          },
          {
            "url": "https://aclanthology.org/2025.acl-long.508.pdf"
          },
          {
            "url": "https://huggingface.co/datasets/callanwu/WebWalkerQA"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[14].evidence",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/WebAgent"
          },
          {
            "url": "https://arxiv.org/abs/2501.07572"
          },
          {
            "url": "https://aclanthology.org/2025.acl-long.508.pdf"
          },
          {
            "url": "https://huggingface.co/datasets/callanwu/WebWalkerQA"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[15].name",
        "citations": [
          {
            "url": "https://github.com/xbench-ai/xbench-evals"
          },
          {
            "url": "https://xbench.org/agi/aisearch"
          },
          {
            "url": "https://xbench.org/files/Eval%20Card%20xbench-DeepSearch.pdf"
          },
          {
            "url": "https://github.com/xbench-ai/xbench-evals/blob/main/data/DeepSearch-2510.csv"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[15].canonical_url",
        "citations": [
          {
            "url": "https://github.com/xbench-ai/xbench-evals"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[15].category",
        "citations": [
          {
            "url": "https://github.com/xbench-ai/xbench-evals"
          },
          {
            "url": "https://xbench.org/agi/aisearch"
          },
          {
            "url": "https://xbench.org/files/Eval%20Card%20xbench-DeepSearch.pdf"
          },
          {
            "url": "https://github.com/xbench-ai/xbench-evals/blob/main/data/DeepSearch-2510.csv"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[15].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/xbench-ai/xbench-evals"
          },
          {
            "url": "https://xbench.org/agi/aisearch"
          },
          {
            "url": "https://xbench.org/files/Eval%20Card%20xbench-DeepSearch.pdf"
          },
          {
            "url": "https://github.com/xbench-ai/xbench-evals/blob/main/data/DeepSearch-2510.csv"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[15].measures",
        "citations": [
          {
            "url": "https://github.com/xbench-ai/xbench-evals"
          },
          {
            "url": "https://xbench.org/agi/aisearch"
          },
          {
            "url": "https://xbench.org/files/Eval%20Card%20xbench-DeepSearch.pdf"
          },
          {
            "url": "https://github.com/xbench-ai/xbench-evals/blob/main/data/DeepSearch-2510.csv"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[15].task_data",
        "citations": [
          {
            "url": "https://github.com/xbench-ai/xbench-evals"
          },
          {
            "url": "https://xbench.org/agi/aisearch"
          },
          {
            "url": "https://xbench.org/files/Eval%20Card%20xbench-DeepSearch.pdf"
          },
          {
            "url": "https://github.com/xbench-ai/xbench-evals/blob/main/data/DeepSearch-2510.csv"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[15].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/xbench-ai/xbench-evals"
          },
          {
            "url": "https://xbench.org/agi/aisearch"
          },
          {
            "url": "https://xbench.org/files/Eval%20Card%20xbench-DeepSearch.pdf"
          },
          {
            "url": "https://github.com/xbench-ai/xbench-evals/blob/main/data/DeepSearch-2510.csv"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[15].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/xbench-ai/xbench-evals"
          },
          {
            "url": "https://xbench.org/agi/aisearch"
          },
          {
            "url": "https://xbench.org/files/Eval%20Card%20xbench-DeepSearch.pdf"
          },
          {
            "url": "https://github.com/xbench-ai/xbench-evals/blob/main/data/DeepSearch-2510.csv"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[15].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/xbench-ai/xbench-evals"
          },
          {
            "url": "https://xbench.org/agi/aisearch"
          },
          {
            "url": "https://xbench.org/files/Eval%20Card%20xbench-DeepSearch.pdf"
          },
          {
            "url": "https://github.com/xbench-ai/xbench-evals/blob/main/data/DeepSearch-2510.csv"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[15].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/xbench-ai/xbench-evals"
          },
          {
            "url": "https://xbench.org/agi/aisearch"
          },
          {
            "url": "https://xbench.org/files/Eval%20Card%20xbench-DeepSearch.pdf"
          },
          {
            "url": "https://github.com/xbench-ai/xbench-evals/blob/main/data/DeepSearch-2510.csv"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[15].evidence",
        "citations": [
          {
            "url": "https://github.com/xbench-ai/xbench-evals"
          },
          {
            "url": "https://xbench.org/agi/aisearch"
          },
          {
            "url": "https://xbench.org/files/Eval%20Card%20xbench-DeepSearch.pdf"
          },
          {
            "url": "https://github.com/xbench-ai/xbench-evals/blob/main/data/DeepSearch-2510.csv"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[16].name",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/vtllms/sealqa"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01062"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[16].canonical_url",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/vtllms/sealqa"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[16].category",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/vtllms/sealqa"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01062"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[16].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/vtllms/sealqa"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01062"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[16].measures",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/vtllms/sealqa"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01062"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[16].task_data",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/vtllms/sealqa"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01062"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[16].evaluation_code",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/vtllms/sealqa"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01062"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[16].judging_criteria",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/vtllms/sealqa"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01062"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[16].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/vtllms/sealqa"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01062"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[16].reported_results_provenance",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/vtllms/sealqa"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01062"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[16].evidence",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/vtllms/sealqa"
          },
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01062"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[17].name",
        "citations": [
          {
            "url": "https://osu-nlp-group.github.io/Mind2Web-2/"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2/blob/main/run_eval.py"
          },
          {
            "url": "https://huggingface.co/datasets/osunlp/Mind2Web-2"
          },
          {
            "url": "https://arxiv.org/html/2506.21506v2"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[17].canonical_url",
        "citations": [
          {
            "url": "https://osu-nlp-group.github.io/Mind2Web-2/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[17].category",
        "citations": [
          {
            "url": "https://osu-nlp-group.github.io/Mind2Web-2/"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2/blob/main/run_eval.py"
          },
          {
            "url": "https://huggingface.co/datasets/osunlp/Mind2Web-2"
          },
          {
            "url": "https://arxiv.org/html/2506.21506v2"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[17].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://osu-nlp-group.github.io/Mind2Web-2/"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2/blob/main/run_eval.py"
          },
          {
            "url": "https://huggingface.co/datasets/osunlp/Mind2Web-2"
          },
          {
            "url": "https://arxiv.org/html/2506.21506v2"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[17].measures",
        "citations": [
          {
            "url": "https://osu-nlp-group.github.io/Mind2Web-2/"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2/blob/main/run_eval.py"
          },
          {
            "url": "https://huggingface.co/datasets/osunlp/Mind2Web-2"
          },
          {
            "url": "https://arxiv.org/html/2506.21506v2"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[17].task_data",
        "citations": [
          {
            "url": "https://osu-nlp-group.github.io/Mind2Web-2/"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2/blob/main/run_eval.py"
          },
          {
            "url": "https://huggingface.co/datasets/osunlp/Mind2Web-2"
          },
          {
            "url": "https://arxiv.org/html/2506.21506v2"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[17].evaluation_code",
        "citations": [
          {
            "url": "https://osu-nlp-group.github.io/Mind2Web-2/"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2/blob/main/run_eval.py"
          },
          {
            "url": "https://huggingface.co/datasets/osunlp/Mind2Web-2"
          },
          {
            "url": "https://arxiv.org/html/2506.21506v2"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[17].judging_criteria",
        "citations": [
          {
            "url": "https://osu-nlp-group.github.io/Mind2Web-2/"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2/blob/main/run_eval.py"
          },
          {
            "url": "https://huggingface.co/datasets/osunlp/Mind2Web-2"
          },
          {
            "url": "https://arxiv.org/html/2506.21506v2"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[17].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://osu-nlp-group.github.io/Mind2Web-2/"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2/blob/main/run_eval.py"
          },
          {
            "url": "https://huggingface.co/datasets/osunlp/Mind2Web-2"
          },
          {
            "url": "https://arxiv.org/html/2506.21506v2"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[17].reported_results_provenance",
        "citations": [
          {
            "url": "https://osu-nlp-group.github.io/Mind2Web-2/"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2/blob/main/run_eval.py"
          },
          {
            "url": "https://huggingface.co/datasets/osunlp/Mind2Web-2"
          },
          {
            "url": "https://arxiv.org/html/2506.21506v2"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[17].evidence",
        "citations": [
          {
            "url": "https://osu-nlp-group.github.io/Mind2Web-2/"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2"
          },
          {
            "url": "https://github.com/OSU-NLP-Group/Mind2Web-2/blob/main/run_eval.py"
          },
          {
            "url": "https://huggingface.co/datasets/osunlp/Mind2Web-2"
          },
          {
            "url": "https://arxiv.org/html/2506.21506v2"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[18].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.06287"
          },
          {
            "url": "https://arxiv.org/html/2506.06287v1"
          },
          {
            "url": "https://futuresearch.ai/deep-research-bench/"
          },
          {
            "url": "https://drb.futuresearch.ai/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[18].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.06287"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[18].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.06287"
          },
          {
            "url": "https://arxiv.org/html/2506.06287v1"
          },
          {
            "url": "https://futuresearch.ai/deep-research-bench/"
          },
          {
            "url": "https://drb.futuresearch.ai/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[18].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.06287"
          },
          {
            "url": "https://arxiv.org/html/2506.06287v1"
          },
          {
            "url": "https://futuresearch.ai/deep-research-bench/"
          },
          {
            "url": "https://drb.futuresearch.ai/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[18].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.06287"
          },
          {
            "url": "https://arxiv.org/html/2506.06287v1"
          },
          {
            "url": "https://futuresearch.ai/deep-research-bench/"
          },
          {
            "url": "https://drb.futuresearch.ai/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[18].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.06287"
          },
          {
            "url": "https://arxiv.org/html/2506.06287v1"
          },
          {
            "url": "https://futuresearch.ai/deep-research-bench/"
          },
          {
            "url": "https://drb.futuresearch.ai/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[18].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.06287"
          },
          {
            "url": "https://arxiv.org/html/2506.06287v1"
          },
          {
            "url": "https://futuresearch.ai/deep-research-bench/"
          },
          {
            "url": "https://drb.futuresearch.ai/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[18].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.06287"
          },
          {
            "url": "https://arxiv.org/html/2506.06287v1"
          },
          {
            "url": "https://futuresearch.ai/deep-research-bench/"
          },
          {
            "url": "https://drb.futuresearch.ai/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[18].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.06287"
          },
          {
            "url": "https://arxiv.org/html/2506.06287v1"
          },
          {
            "url": "https://futuresearch.ai/deep-research-bench/"
          },
          {
            "url": "https://drb.futuresearch.ai/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[18].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.06287"
          },
          {
            "url": "https://arxiv.org/html/2506.06287v1"
          },
          {
            "url": "https://futuresearch.ai/deep-research-bench/"
          },
          {
            "url": "https://drb.futuresearch.ai/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[18].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.06287"
          },
          {
            "url": "https://arxiv.org/html/2506.06287v1"
          },
          {
            "url": "https://futuresearch.ai/deep-research-bench/"
          },
          {
            "url": "https://drb.futuresearch.ai/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[19].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2601.20975"
          },
          {
            "url": "https://arxiv.org/html/2601.20975v1"
          },
          {
            "url": "https://huggingface.co/datasets/google/deepsearchqa"
          },
          {
            "url": "https://www.kaggle.com/benchmarks/google/dsqa/leaderboard"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[19].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2601.20975"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[19].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2601.20975"
          },
          {
            "url": "https://arxiv.org/html/2601.20975v1"
          },
          {
            "url": "https://huggingface.co/datasets/google/deepsearchqa"
          },
          {
            "url": "https://www.kaggle.com/benchmarks/google/dsqa/leaderboard"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[19].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2601.20975"
          },
          {
            "url": "https://arxiv.org/html/2601.20975v1"
          },
          {
            "url": "https://huggingface.co/datasets/google/deepsearchqa"
          },
          {
            "url": "https://www.kaggle.com/benchmarks/google/dsqa/leaderboard"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[19].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2601.20975"
          },
          {
            "url": "https://arxiv.org/html/2601.20975v1"
          },
          {
            "url": "https://huggingface.co/datasets/google/deepsearchqa"
          },
          {
            "url": "https://www.kaggle.com/benchmarks/google/dsqa/leaderboard"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[19].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2601.20975"
          },
          {
            "url": "https://arxiv.org/html/2601.20975v1"
          },
          {
            "url": "https://huggingface.co/datasets/google/deepsearchqa"
          },
          {
            "url": "https://www.kaggle.com/benchmarks/google/dsqa/leaderboard"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[19].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2601.20975"
          },
          {
            "url": "https://arxiv.org/html/2601.20975v1"
          },
          {
            "url": "https://huggingface.co/datasets/google/deepsearchqa"
          },
          {
            "url": "https://www.kaggle.com/benchmarks/google/dsqa/leaderboard"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[19].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2601.20975"
          },
          {
            "url": "https://arxiv.org/html/2601.20975v1"
          },
          {
            "url": "https://huggingface.co/datasets/google/deepsearchqa"
          },
          {
            "url": "https://www.kaggle.com/benchmarks/google/dsqa/leaderboard"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[19].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2601.20975"
          },
          {
            "url": "https://arxiv.org/html/2601.20975v1"
          },
          {
            "url": "https://huggingface.co/datasets/google/deepsearchqa"
          },
          {
            "url": "https://www.kaggle.com/benchmarks/google/dsqa/leaderboard"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[19].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2601.20975"
          },
          {
            "url": "https://arxiv.org/html/2601.20975v1"
          },
          {
            "url": "https://huggingface.co/datasets/google/deepsearchqa"
          },
          {
            "url": "https://www.kaggle.com/benchmarks/google/dsqa/leaderboard"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[19].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2601.20975"
          },
          {
            "url": "https://arxiv.org/html/2601.20975v1"
          },
          {
            "url": "https://huggingface.co/datasets/google/deepsearchqa"
          },
          {
            "url": "https://www.kaggle.com/benchmarks/google/dsqa/leaderboard"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[20].name",
        "citations": [
          {
            "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark"
          },
          {
            "url": "https://arxiv.org/html/2602.11685v1"
          },
          {
            "url": "https://hf.co/datasets/perplexity-ai/draco"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[20].canonical_url",
        "citations": [
          {
            "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[20].category",
        "citations": [
          {
            "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark"
          },
          {
            "url": "https://arxiv.org/html/2602.11685v1"
          },
          {
            "url": "https://hf.co/datasets/perplexity-ai/draco"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[20].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark"
          },
          {
            "url": "https://arxiv.org/html/2602.11685v1"
          },
          {
            "url": "https://hf.co/datasets/perplexity-ai/draco"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[20].measures",
        "citations": [
          {
            "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark"
          },
          {
            "url": "https://arxiv.org/html/2602.11685v1"
          },
          {
            "url": "https://hf.co/datasets/perplexity-ai/draco"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[20].task_data",
        "citations": [
          {
            "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark"
          },
          {
            "url": "https://arxiv.org/html/2602.11685v1"
          },
          {
            "url": "https://hf.co/datasets/perplexity-ai/draco"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[20].evaluation_code",
        "citations": [
          {
            "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark"
          },
          {
            "url": "https://arxiv.org/html/2602.11685v1"
          },
          {
            "url": "https://hf.co/datasets/perplexity-ai/draco"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[20].judging_criteria",
        "citations": [
          {
            "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark"
          },
          {
            "url": "https://arxiv.org/html/2602.11685v1"
          },
          {
            "url": "https://hf.co/datasets/perplexity-ai/draco"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[20].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark"
          },
          {
            "url": "https://arxiv.org/html/2602.11685v1"
          },
          {
            "url": "https://hf.co/datasets/perplexity-ai/draco"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[20].reported_results_provenance",
        "citations": [
          {
            "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark"
          },
          {
            "url": "https://arxiv.org/html/2602.11685v1"
          },
          {
            "url": "https://hf.co/datasets/perplexity-ai/draco"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[20].evidence",
        "citations": [
          {
            "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark"
          },
          {
            "url": "https://arxiv.org/html/2602.11685v1"
          },
          {
            "url": "https://hf.co/datasets/perplexity-ai/draco"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[21].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.14747"
          },
          {
            "url": "https://arxiv.org/html/2608.14747"
          },
          {
            "url": "https://github.com/perplexityai/wandr"
          },
          {
            "url": "https://research.perplexity.ai/articles/wandr-benchmark-evaluating-research-agents-that-must-search-wide-and-deep"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[21].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.14747"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[21].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.14747"
          },
          {
            "url": "https://arxiv.org/html/2608.14747"
          },
          {
            "url": "https://github.com/perplexityai/wandr"
          },
          {
            "url": "https://research.perplexity.ai/articles/wandr-benchmark-evaluating-research-agents-that-must-search-wide-and-deep"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[21].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.14747"
          },
          {
            "url": "https://arxiv.org/html/2608.14747"
          },
          {
            "url": "https://github.com/perplexityai/wandr"
          },
          {
            "url": "https://research.perplexity.ai/articles/wandr-benchmark-evaluating-research-agents-that-must-search-wide-and-deep"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[21].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.14747"
          },
          {
            "url": "https://arxiv.org/html/2608.14747"
          },
          {
            "url": "https://github.com/perplexityai/wandr"
          },
          {
            "url": "https://research.perplexity.ai/articles/wandr-benchmark-evaluating-research-agents-that-must-search-wide-and-deep"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[21].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.14747"
          },
          {
            "url": "https://arxiv.org/html/2608.14747"
          },
          {
            "url": "https://github.com/perplexityai/wandr"
          },
          {
            "url": "https://research.perplexity.ai/articles/wandr-benchmark-evaluating-research-agents-that-must-search-wide-and-deep"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[21].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.14747"
          },
          {
            "url": "https://arxiv.org/html/2608.14747"
          },
          {
            "url": "https://github.com/perplexityai/wandr"
          },
          {
            "url": "https://research.perplexity.ai/articles/wandr-benchmark-evaluating-research-agents-that-must-search-wide-and-deep"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[21].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.14747"
          },
          {
            "url": "https://arxiv.org/html/2608.14747"
          },
          {
            "url": "https://github.com/perplexityai/wandr"
          },
          {
            "url": "https://research.perplexity.ai/articles/wandr-benchmark-evaluating-research-agents-that-must-search-wide-and-deep"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[21].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.14747"
          },
          {
            "url": "https://arxiv.org/html/2608.14747"
          },
          {
            "url": "https://github.com/perplexityai/wandr"
          },
          {
            "url": "https://research.perplexity.ai/articles/wandr-benchmark-evaluating-research-agents-that-must-search-wide-and-deep"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[21].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.14747"
          },
          {
            "url": "https://arxiv.org/html/2608.14747"
          },
          {
            "url": "https://github.com/perplexityai/wandr"
          },
          {
            "url": "https://research.perplexity.ai/articles/wandr-benchmark-evaluating-research-agents-that-must-search-wide-and-deep"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[21].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.14747"
          },
          {
            "url": "https://arxiv.org/html/2608.14747"
          },
          {
            "url": "https://github.com/perplexityai/wandr"
          },
          {
            "url": "https://research.perplexity.ai/articles/wandr-benchmark-evaluating-research-agents-that-must-search-wide-and-deep"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[22].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.06724"
          },
          {
            "url": "https://arxiv.org/html/2602.06724"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://arxiv.org/abs/2510.20168"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[22].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.06724"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[22].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.06724"
          },
          {
            "url": "https://arxiv.org/html/2602.06724"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://arxiv.org/abs/2510.20168"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[22].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.06724"
          },
          {
            "url": "https://arxiv.org/html/2602.06724"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://arxiv.org/abs/2510.20168"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[22].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.06724"
          },
          {
            "url": "https://arxiv.org/html/2602.06724"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://arxiv.org/abs/2510.20168"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[22].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.06724"
          },
          {
            "url": "https://arxiv.org/html/2602.06724"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://arxiv.org/abs/2510.20168"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[22].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.06724"
          },
          {
            "url": "https://arxiv.org/html/2602.06724"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://arxiv.org/abs/2510.20168"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[22].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.06724"
          },
          {
            "url": "https://arxiv.org/html/2602.06724"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://arxiv.org/abs/2510.20168"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[22].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.06724"
          },
          {
            "url": "https://arxiv.org/html/2602.06724"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://arxiv.org/abs/2510.20168"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[22].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.06724"
          },
          {
            "url": "https://arxiv.org/html/2602.06724"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://arxiv.org/abs/2510.20168"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[22].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.06724"
          },
          {
            "url": "https://arxiv.org/html/2602.06724"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://arxiv.org/abs/2510.20168"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[23].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.21482"
          },
          {
            "url": "https://huggingface.co/datasets/deepweb-bench-anon/deepweb-bench"
          },
          {
            "url": "https://sixiongxie1001-dot.github.io/deep-research-benchmark2.0"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[23].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.21482"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[23].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.21482"
          },
          {
            "url": "https://huggingface.co/datasets/deepweb-bench-anon/deepweb-bench"
          },
          {
            "url": "https://sixiongxie1001-dot.github.io/deep-research-benchmark2.0"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[23].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.21482"
          },
          {
            "url": "https://huggingface.co/datasets/deepweb-bench-anon/deepweb-bench"
          },
          {
            "url": "https://sixiongxie1001-dot.github.io/deep-research-benchmark2.0"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[23].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.21482"
          },
          {
            "url": "https://huggingface.co/datasets/deepweb-bench-anon/deepweb-bench"
          },
          {
            "url": "https://sixiongxie1001-dot.github.io/deep-research-benchmark2.0"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[23].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.21482"
          },
          {
            "url": "https://huggingface.co/datasets/deepweb-bench-anon/deepweb-bench"
          },
          {
            "url": "https://sixiongxie1001-dot.github.io/deep-research-benchmark2.0"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[23].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.21482"
          },
          {
            "url": "https://huggingface.co/datasets/deepweb-bench-anon/deepweb-bench"
          },
          {
            "url": "https://sixiongxie1001-dot.github.io/deep-research-benchmark2.0"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[23].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.21482"
          },
          {
            "url": "https://huggingface.co/datasets/deepweb-bench-anon/deepweb-bench"
          },
          {
            "url": "https://sixiongxie1001-dot.github.io/deep-research-benchmark2.0"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[23].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.21482"
          },
          {
            "url": "https://huggingface.co/datasets/deepweb-bench-anon/deepweb-bench"
          },
          {
            "url": "https://sixiongxie1001-dot.github.io/deep-research-benchmark2.0"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[23].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.21482"
          },
          {
            "url": "https://huggingface.co/datasets/deepweb-bench-anon/deepweb-bench"
          },
          {
            "url": "https://sixiongxie1001-dot.github.io/deep-research-benchmark2.0"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[23].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.21482"
          },
          {
            "url": "https://huggingface.co/datasets/deepweb-bench-anon/deepweb-bench"
          },
          {
            "url": "https://sixiongxie1001-dot.github.io/deep-research-benchmark2.0"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[24].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.02163"
          },
          {
            "url": "https://arxiv.org/html/2608.02163v1"
          },
          {
            "url": "https://github.com/chr6192/TaskEvolving.git"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[24].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.02163"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[24].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.02163"
          },
          {
            "url": "https://arxiv.org/html/2608.02163v1"
          },
          {
            "url": "https://github.com/chr6192/TaskEvolving.git"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[24].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.02163"
          },
          {
            "url": "https://arxiv.org/html/2608.02163v1"
          },
          {
            "url": "https://github.com/chr6192/TaskEvolving.git"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[24].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.02163"
          },
          {
            "url": "https://arxiv.org/html/2608.02163v1"
          },
          {
            "url": "https://github.com/chr6192/TaskEvolving.git"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[24].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.02163"
          },
          {
            "url": "https://arxiv.org/html/2608.02163v1"
          },
          {
            "url": "https://github.com/chr6192/TaskEvolving.git"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[24].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.02163"
          },
          {
            "url": "https://arxiv.org/html/2608.02163v1"
          },
          {
            "url": "https://github.com/chr6192/TaskEvolving.git"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[24].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.02163"
          },
          {
            "url": "https://arxiv.org/html/2608.02163v1"
          },
          {
            "url": "https://github.com/chr6192/TaskEvolving.git"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[24].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.02163"
          },
          {
            "url": "https://arxiv.org/html/2608.02163v1"
          },
          {
            "url": "https://github.com/chr6192/TaskEvolving.git"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[24].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.02163"
          },
          {
            "url": "https://arxiv.org/html/2608.02163v1"
          },
          {
            "url": "https://github.com/chr6192/TaskEvolving.git"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[24].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2608.02163"
          },
          {
            "url": "https://arxiv.org/html/2608.02163v1"
          },
          {
            "url": "https://github.com/chr6192/TaskEvolving.git"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[25].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.01152"
          },
          {
            "url": "https://arxiv.org/html/2603.01152v2"
          },
          {
            "url": "https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1"
          },
          {
            "url": "https://huggingface.co/datasets/artillerywu/DeepResearch-9K"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[25].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.01152"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[25].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.01152"
          },
          {
            "url": "https://arxiv.org/html/2603.01152v2"
          },
          {
            "url": "https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1"
          },
          {
            "url": "https://huggingface.co/datasets/artillerywu/DeepResearch-9K"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[25].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.01152"
          },
          {
            "url": "https://arxiv.org/html/2603.01152v2"
          },
          {
            "url": "https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1"
          },
          {
            "url": "https://huggingface.co/datasets/artillerywu/DeepResearch-9K"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[25].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.01152"
          },
          {
            "url": "https://arxiv.org/html/2603.01152v2"
          },
          {
            "url": "https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1"
          },
          {
            "url": "https://huggingface.co/datasets/artillerywu/DeepResearch-9K"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[25].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.01152"
          },
          {
            "url": "https://arxiv.org/html/2603.01152v2"
          },
          {
            "url": "https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1"
          },
          {
            "url": "https://huggingface.co/datasets/artillerywu/DeepResearch-9K"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[25].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.01152"
          },
          {
            "url": "https://arxiv.org/html/2603.01152v2"
          },
          {
            "url": "https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1"
          },
          {
            "url": "https://huggingface.co/datasets/artillerywu/DeepResearch-9K"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[25].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.01152"
          },
          {
            "url": "https://arxiv.org/html/2603.01152v2"
          },
          {
            "url": "https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1"
          },
          {
            "url": "https://huggingface.co/datasets/artillerywu/DeepResearch-9K"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[25].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.01152"
          },
          {
            "url": "https://arxiv.org/html/2603.01152v2"
          },
          {
            "url": "https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1"
          },
          {
            "url": "https://huggingface.co/datasets/artillerywu/DeepResearch-9K"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[25].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.01152"
          },
          {
            "url": "https://arxiv.org/html/2603.01152v2"
          },
          {
            "url": "https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1"
          },
          {
            "url": "https://huggingface.co/datasets/artillerywu/DeepResearch-9K"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[25].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.01152"
          },
          {
            "url": "https://arxiv.org/html/2603.01152v2"
          },
          {
            "url": "https://github.com/Applied-Machine-Learning-Lab/SIGIR2026_DeepResearch-R1"
          },
          {
            "url": "https://huggingface.co/datasets/artillerywu/DeepResearch-9K"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[26].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2609.11318"
          },
          {
            "url": "https://arxiv.org/html/2609.11318v1"
          },
          {
            "url": "https://github.com/minghaoguo20/Mr-LHDR-eval"
          },
          {
            "url": "https://huggingface.co/datasets/Henryeahhh/Mr-LHDR"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[26].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2609.11318"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[26].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2609.11318"
          },
          {
            "url": "https://arxiv.org/html/2609.11318v1"
          },
          {
            "url": "https://github.com/minghaoguo20/Mr-LHDR-eval"
          },
          {
            "url": "https://huggingface.co/datasets/Henryeahhh/Mr-LHDR"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[26].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2609.11318"
          },
          {
            "url": "https://arxiv.org/html/2609.11318v1"
          },
          {
            "url": "https://github.com/minghaoguo20/Mr-LHDR-eval"
          },
          {
            "url": "https://huggingface.co/datasets/Henryeahhh/Mr-LHDR"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[26].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2609.11318"
          },
          {
            "url": "https://arxiv.org/html/2609.11318v1"
          },
          {
            "url": "https://github.com/minghaoguo20/Mr-LHDR-eval"
          },
          {
            "url": "https://huggingface.co/datasets/Henryeahhh/Mr-LHDR"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[26].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2609.11318"
          },
          {
            "url": "https://arxiv.org/html/2609.11318v1"
          },
          {
            "url": "https://github.com/minghaoguo20/Mr-LHDR-eval"
          },
          {
            "url": "https://huggingface.co/datasets/Henryeahhh/Mr-LHDR"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[26].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2609.11318"
          },
          {
            "url": "https://arxiv.org/html/2609.11318v1"
          },
          {
            "url": "https://github.com/minghaoguo20/Mr-LHDR-eval"
          },
          {
            "url": "https://huggingface.co/datasets/Henryeahhh/Mr-LHDR"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[26].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2609.11318"
          },
          {
            "url": "https://arxiv.org/html/2609.11318v1"
          },
          {
            "url": "https://github.com/minghaoguo20/Mr-LHDR-eval"
          },
          {
            "url": "https://huggingface.co/datasets/Henryeahhh/Mr-LHDR"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[26].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2609.11318"
          },
          {
            "url": "https://arxiv.org/html/2609.11318v1"
          },
          {
            "url": "https://github.com/minghaoguo20/Mr-LHDR-eval"
          },
          {
            "url": "https://huggingface.co/datasets/Henryeahhh/Mr-LHDR"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[26].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2609.11318"
          },
          {
            "url": "https://arxiv.org/html/2609.11318v1"
          },
          {
            "url": "https://github.com/minghaoguo20/Mr-LHDR-eval"
          },
          {
            "url": "https://huggingface.co/datasets/Henryeahhh/Mr-LHDR"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[26].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2609.11318"
          },
          {
            "url": "https://arxiv.org/html/2609.11318v1"
          },
          {
            "url": "https://github.com/minghaoguo20/Mr-LHDR-eval"
          },
          {
            "url": "https://huggingface.co/datasets/Henryeahhh/Mr-LHDR"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[27].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.13160"
          },
          {
            "url": "https://arxiv.org/html/2509.13160"
          },
          {
            "url": "https://github.com/randomtutu/FinSearchComp"
          },
          {
            "url": "https://randomtutu.github.io/FinSearchComp/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[27].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.13160"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[27].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.13160"
          },
          {
            "url": "https://arxiv.org/html/2509.13160"
          },
          {
            "url": "https://github.com/randomtutu/FinSearchComp"
          },
          {
            "url": "https://randomtutu.github.io/FinSearchComp/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[27].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.13160"
          },
          {
            "url": "https://arxiv.org/html/2509.13160"
          },
          {
            "url": "https://github.com/randomtutu/FinSearchComp"
          },
          {
            "url": "https://randomtutu.github.io/FinSearchComp/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[27].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.13160"
          },
          {
            "url": "https://arxiv.org/html/2509.13160"
          },
          {
            "url": "https://github.com/randomtutu/FinSearchComp"
          },
          {
            "url": "https://randomtutu.github.io/FinSearchComp/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[27].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.13160"
          },
          {
            "url": "https://arxiv.org/html/2509.13160"
          },
          {
            "url": "https://github.com/randomtutu/FinSearchComp"
          },
          {
            "url": "https://randomtutu.github.io/FinSearchComp/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[27].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.13160"
          },
          {
            "url": "https://arxiv.org/html/2509.13160"
          },
          {
            "url": "https://github.com/randomtutu/FinSearchComp"
          },
          {
            "url": "https://randomtutu.github.io/FinSearchComp/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[27].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.13160"
          },
          {
            "url": "https://arxiv.org/html/2509.13160"
          },
          {
            "url": "https://github.com/randomtutu/FinSearchComp"
          },
          {
            "url": "https://randomtutu.github.io/FinSearchComp/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[27].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.13160"
          },
          {
            "url": "https://arxiv.org/html/2509.13160"
          },
          {
            "url": "https://github.com/randomtutu/FinSearchComp"
          },
          {
            "url": "https://randomtutu.github.io/FinSearchComp/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[27].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.13160"
          },
          {
            "url": "https://arxiv.org/html/2509.13160"
          },
          {
            "url": "https://github.com/randomtutu/FinSearchComp"
          },
          {
            "url": "https://randomtutu.github.io/FinSearchComp/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[27].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.13160"
          },
          {
            "url": "https://arxiv.org/html/2509.13160"
          },
          {
            "url": "https://github.com/randomtutu/FinSearchComp"
          },
          {
            "url": "https://randomtutu.github.io/FinSearchComp/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[28].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.00828"
          },
          {
            "url": "https://arxiv.org/html/2508.00828"
          },
          {
            "url": "https://github.com/vals-ai/finance-agent"
          },
          {
            "url": "https://doi.org/10.5281/zenodo.15428823"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[28].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.00828"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[28].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.00828"
          },
          {
            "url": "https://arxiv.org/html/2508.00828"
          },
          {
            "url": "https://github.com/vals-ai/finance-agent"
          },
          {
            "url": "https://doi.org/10.5281/zenodo.15428823"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[28].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.00828"
          },
          {
            "url": "https://arxiv.org/html/2508.00828"
          },
          {
            "url": "https://github.com/vals-ai/finance-agent"
          },
          {
            "url": "https://doi.org/10.5281/zenodo.15428823"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[28].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.00828"
          },
          {
            "url": "https://arxiv.org/html/2508.00828"
          },
          {
            "url": "https://github.com/vals-ai/finance-agent"
          },
          {
            "url": "https://doi.org/10.5281/zenodo.15428823"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[28].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.00828"
          },
          {
            "url": "https://arxiv.org/html/2508.00828"
          },
          {
            "url": "https://github.com/vals-ai/finance-agent"
          },
          {
            "url": "https://doi.org/10.5281/zenodo.15428823"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[28].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.00828"
          },
          {
            "url": "https://arxiv.org/html/2508.00828"
          },
          {
            "url": "https://github.com/vals-ai/finance-agent"
          },
          {
            "url": "https://doi.org/10.5281/zenodo.15428823"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[28].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.00828"
          },
          {
            "url": "https://arxiv.org/html/2508.00828"
          },
          {
            "url": "https://github.com/vals-ai/finance-agent"
          },
          {
            "url": "https://doi.org/10.5281/zenodo.15428823"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[28].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.00828"
          },
          {
            "url": "https://arxiv.org/html/2508.00828"
          },
          {
            "url": "https://github.com/vals-ai/finance-agent"
          },
          {
            "url": "https://doi.org/10.5281/zenodo.15428823"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[28].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.00828"
          },
          {
            "url": "https://arxiv.org/html/2508.00828"
          },
          {
            "url": "https://github.com/vals-ai/finance-agent"
          },
          {
            "url": "https://doi.org/10.5281/zenodo.15428823"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[28].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.00828"
          },
          {
            "url": "https://arxiv.org/html/2508.00828"
          },
          {
            "url": "https://github.com/vals-ai/finance-agent"
          },
          {
            "url": "https://doi.org/10.5281/zenodo.15428823"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[29].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.04403"
          },
          {
            "url": "https://github.com/daloopa/finretrieval"
          },
          {
            "url": "https://huggingface.co/datasets/daloopa/finretrieval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[29].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.04403"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[29].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.04403"
          },
          {
            "url": "https://github.com/daloopa/finretrieval"
          },
          {
            "url": "https://huggingface.co/datasets/daloopa/finretrieval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[29].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.04403"
          },
          {
            "url": "https://github.com/daloopa/finretrieval"
          },
          {
            "url": "https://huggingface.co/datasets/daloopa/finretrieval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[29].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.04403"
          },
          {
            "url": "https://github.com/daloopa/finretrieval"
          },
          {
            "url": "https://huggingface.co/datasets/daloopa/finretrieval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[29].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.04403"
          },
          {
            "url": "https://github.com/daloopa/finretrieval"
          },
          {
            "url": "https://huggingface.co/datasets/daloopa/finretrieval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[29].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.04403"
          },
          {
            "url": "https://github.com/daloopa/finretrieval"
          },
          {
            "url": "https://huggingface.co/datasets/daloopa/finretrieval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[29].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.04403"
          },
          {
            "url": "https://github.com/daloopa/finretrieval"
          },
          {
            "url": "https://huggingface.co/datasets/daloopa/finretrieval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[29].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.04403"
          },
          {
            "url": "https://github.com/daloopa/finretrieval"
          },
          {
            "url": "https://huggingface.co/datasets/daloopa/finretrieval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[29].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.04403"
          },
          {
            "url": "https://github.com/daloopa/finretrieval"
          },
          {
            "url": "https://huggingface.co/datasets/daloopa/finretrieval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[29].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2603.04403"
          },
          {
            "url": "https://github.com/daloopa/finretrieval"
          },
          {
            "url": "https://huggingface.co/datasets/daloopa/finretrieval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[30].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14963"
          },
          {
            "url": "https://arxiv.org/html/2505.14963"
          },
          {
            "url": "https://github.com/shan23chen/MedBrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/AIM-Harvard/MedBrowseComp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[30].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14963"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[30].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14963"
          },
          {
            "url": "https://arxiv.org/html/2505.14963"
          },
          {
            "url": "https://github.com/shan23chen/MedBrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/AIM-Harvard/MedBrowseComp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[30].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14963"
          },
          {
            "url": "https://arxiv.org/html/2505.14963"
          },
          {
            "url": "https://github.com/shan23chen/MedBrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/AIM-Harvard/MedBrowseComp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[30].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14963"
          },
          {
            "url": "https://arxiv.org/html/2505.14963"
          },
          {
            "url": "https://github.com/shan23chen/MedBrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/AIM-Harvard/MedBrowseComp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[30].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14963"
          },
          {
            "url": "https://arxiv.org/html/2505.14963"
          },
          {
            "url": "https://github.com/shan23chen/MedBrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/AIM-Harvard/MedBrowseComp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[30].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14963"
          },
          {
            "url": "https://arxiv.org/html/2505.14963"
          },
          {
            "url": "https://github.com/shan23chen/MedBrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/AIM-Harvard/MedBrowseComp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[30].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14963"
          },
          {
            "url": "https://arxiv.org/html/2505.14963"
          },
          {
            "url": "https://github.com/shan23chen/MedBrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/AIM-Harvard/MedBrowseComp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[30].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14963"
          },
          {
            "url": "https://arxiv.org/html/2505.14963"
          },
          {
            "url": "https://github.com/shan23chen/MedBrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/AIM-Harvard/MedBrowseComp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[30].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14963"
          },
          {
            "url": "https://arxiv.org/html/2505.14963"
          },
          {
            "url": "https://github.com/shan23chen/MedBrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/AIM-Harvard/MedBrowseComp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[30].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14963"
          },
          {
            "url": "https://arxiv.org/html/2505.14963"
          },
          {
            "url": "https://github.com/shan23chen/MedBrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/AIM-Harvard/MedBrowseComp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[31].name",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.acl-long.572/"
          },
          {
            "url": "https://github.com/bytedance/pasa"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[31].canonical_url",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.acl-long.572/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[31].category",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.acl-long.572/"
          },
          {
            "url": "https://github.com/bytedance/pasa"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[31].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.acl-long.572/"
          },
          {
            "url": "https://github.com/bytedance/pasa"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[31].measures",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.acl-long.572/"
          },
          {
            "url": "https://github.com/bytedance/pasa"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[31].task_data",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.acl-long.572/"
          },
          {
            "url": "https://github.com/bytedance/pasa"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[31].evaluation_code",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.acl-long.572/"
          },
          {
            "url": "https://github.com/bytedance/pasa"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[31].judging_criteria",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.acl-long.572/"
          },
          {
            "url": "https://github.com/bytedance/pasa"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[31].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.acl-long.572/"
          },
          {
            "url": "https://github.com/bytedance/pasa"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[31].reported_results_provenance",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.acl-long.572/"
          },
          {
            "url": "https://github.com/bytedance/pasa"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[31].evidence",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.acl-long.572/"
          },
          {
            "url": "https://github.com/bytedance/pasa"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[32].name",
        "citations": [
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/openbenchmarks-labs/multi-turn-company-search"
          },
          {
            "url": "https://huggingface.co/datasets/openbenchmarks/OB-Company-Websearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[32].canonical_url",
        "citations": [
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[32].category",
        "citations": [
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/openbenchmarks-labs/multi-turn-company-search"
          },
          {
            "url": "https://huggingface.co/datasets/openbenchmarks/OB-Company-Websearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[32].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/openbenchmarks-labs/multi-turn-company-search"
          },
          {
            "url": "https://huggingface.co/datasets/openbenchmarks/OB-Company-Websearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[32].measures",
        "citations": [
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/openbenchmarks-labs/multi-turn-company-search"
          },
          {
            "url": "https://huggingface.co/datasets/openbenchmarks/OB-Company-Websearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[32].task_data",
        "citations": [
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/openbenchmarks-labs/multi-turn-company-search"
          },
          {
            "url": "https://huggingface.co/datasets/openbenchmarks/OB-Company-Websearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[32].evaluation_code",
        "citations": [
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/openbenchmarks-labs/multi-turn-company-search"
          },
          {
            "url": "https://huggingface.co/datasets/openbenchmarks/OB-Company-Websearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[32].judging_criteria",
        "citations": [
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/openbenchmarks-labs/multi-turn-company-search"
          },
          {
            "url": "https://huggingface.co/datasets/openbenchmarks/OB-Company-Websearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[32].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/openbenchmarks-labs/multi-turn-company-search"
          },
          {
            "url": "https://huggingface.co/datasets/openbenchmarks/OB-Company-Websearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[32].reported_results_provenance",
        "citations": [
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/openbenchmarks-labs/multi-turn-company-search"
          },
          {
            "url": "https://huggingface.co/datasets/openbenchmarks/OB-Company-Websearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[32].evidence",
        "citations": [
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/openbenchmarks-labs/multi-turn-company-search"
          },
          {
            "url": "https://huggingface.co/datasets/openbenchmarks/OB-Company-Websearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[33].name",
        "citations": [
          {
            "url": "https://github.com/MMBrowseComp/MM-BrowseComp"
          },
          {
            "url": "https://arxiv.org/abs/2508.13186"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[33].canonical_url",
        "citations": [
          {
            "url": "https://github.com/MMBrowseComp/MM-BrowseComp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[33].category",
        "citations": [
          {
            "url": "https://github.com/MMBrowseComp/MM-BrowseComp"
          },
          {
            "url": "https://arxiv.org/abs/2508.13186"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[33].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/MMBrowseComp/MM-BrowseComp"
          },
          {
            "url": "https://arxiv.org/abs/2508.13186"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[33].measures",
        "citations": [
          {
            "url": "https://github.com/MMBrowseComp/MM-BrowseComp"
          },
          {
            "url": "https://arxiv.org/abs/2508.13186"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[33].task_data",
        "citations": [
          {
            "url": "https://github.com/MMBrowseComp/MM-BrowseComp"
          },
          {
            "url": "https://arxiv.org/abs/2508.13186"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[33].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/MMBrowseComp/MM-BrowseComp"
          },
          {
            "url": "https://arxiv.org/abs/2508.13186"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[33].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/MMBrowseComp/MM-BrowseComp"
          },
          {
            "url": "https://arxiv.org/abs/2508.13186"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[33].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/MMBrowseComp/MM-BrowseComp"
          },
          {
            "url": "https://arxiv.org/abs/2508.13186"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[33].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/MMBrowseComp/MM-BrowseComp"
          },
          {
            "url": "https://arxiv.org/abs/2508.13186"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[33].evidence",
        "citations": [
          {
            "url": "https://github.com/MMBrowseComp/MM-BrowseComp"
          },
          {
            "url": "https://arxiv.org/abs/2508.13186"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[34].name",
        "citations": [
          {
            "url": "https://github.com/mmsearch-plus/MMSearch-Plus"
          },
          {
            "url": "https://arxiv.org/abs/2508.21475"
          },
          {
            "url": "https://huggingface.co/datasets/Cie1/MMSearch-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[34].canonical_url",
        "citations": [
          {
            "url": "https://github.com/mmsearch-plus/MMSearch-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[34].category",
        "citations": [
          {
            "url": "https://github.com/mmsearch-plus/MMSearch-Plus"
          },
          {
            "url": "https://arxiv.org/abs/2508.21475"
          },
          {
            "url": "https://huggingface.co/datasets/Cie1/MMSearch-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[34].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/mmsearch-plus/MMSearch-Plus"
          },
          {
            "url": "https://arxiv.org/abs/2508.21475"
          },
          {
            "url": "https://huggingface.co/datasets/Cie1/MMSearch-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[34].measures",
        "citations": [
          {
            "url": "https://github.com/mmsearch-plus/MMSearch-Plus"
          },
          {
            "url": "https://arxiv.org/abs/2508.21475"
          },
          {
            "url": "https://huggingface.co/datasets/Cie1/MMSearch-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[34].task_data",
        "citations": [
          {
            "url": "https://github.com/mmsearch-plus/MMSearch-Plus"
          },
          {
            "url": "https://arxiv.org/abs/2508.21475"
          },
          {
            "url": "https://huggingface.co/datasets/Cie1/MMSearch-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[34].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/mmsearch-plus/MMSearch-Plus"
          },
          {
            "url": "https://arxiv.org/abs/2508.21475"
          },
          {
            "url": "https://huggingface.co/datasets/Cie1/MMSearch-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[34].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/mmsearch-plus/MMSearch-Plus"
          },
          {
            "url": "https://arxiv.org/abs/2508.21475"
          },
          {
            "url": "https://huggingface.co/datasets/Cie1/MMSearch-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[34].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/mmsearch-plus/MMSearch-Plus"
          },
          {
            "url": "https://arxiv.org/abs/2508.21475"
          },
          {
            "url": "https://huggingface.co/datasets/Cie1/MMSearch-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[34].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/mmsearch-plus/MMSearch-Plus"
          },
          {
            "url": "https://arxiv.org/abs/2508.21475"
          },
          {
            "url": "https://huggingface.co/datasets/Cie1/MMSearch-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[34].evidence",
        "citations": [
          {
            "url": "https://github.com/mmsearch-plus/MMSearch-Plus"
          },
          {
            "url": "https://arxiv.org/abs/2508.21475"
          },
          {
            "url": "https://huggingface.co/datasets/Cie1/MMSearch-Plus"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[35].name",
        "citations": [
          {
            "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3"
          },
          {
            "url": "https://arxiv.org/abs/2602.12876"
          },
          {
            "url": "https://halcyon-zhang.github.io/BrowseComp-V3/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[35].canonical_url",
        "citations": [
          {
            "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[35].category",
        "citations": [
          {
            "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3"
          },
          {
            "url": "https://arxiv.org/abs/2602.12876"
          },
          {
            "url": "https://halcyon-zhang.github.io/BrowseComp-V3/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[35].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3"
          },
          {
            "url": "https://arxiv.org/abs/2602.12876"
          },
          {
            "url": "https://halcyon-zhang.github.io/BrowseComp-V3/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[35].measures",
        "citations": [
          {
            "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3"
          },
          {
            "url": "https://arxiv.org/abs/2602.12876"
          },
          {
            "url": "https://halcyon-zhang.github.io/BrowseComp-V3/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[35].task_data",
        "citations": [
          {
            "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3"
          },
          {
            "url": "https://arxiv.org/abs/2602.12876"
          },
          {
            "url": "https://halcyon-zhang.github.io/BrowseComp-V3/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[35].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3"
          },
          {
            "url": "https://arxiv.org/abs/2602.12876"
          },
          {
            "url": "https://halcyon-zhang.github.io/BrowseComp-V3/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[35].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3"
          },
          {
            "url": "https://arxiv.org/abs/2602.12876"
          },
          {
            "url": "https://halcyon-zhang.github.io/BrowseComp-V3/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[35].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3"
          },
          {
            "url": "https://arxiv.org/abs/2602.12876"
          },
          {
            "url": "https://halcyon-zhang.github.io/BrowseComp-V3/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[35].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3"
          },
          {
            "url": "https://arxiv.org/abs/2602.12876"
          },
          {
            "url": "https://halcyon-zhang.github.io/BrowseComp-V3/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[35].evidence",
        "citations": [
          {
            "url": "https://github.com/Halcyon-Zhang/BrowseComp-V3"
          },
          {
            "url": "https://arxiv.org/abs/2602.12876"
          },
          {
            "url": "https://halcyon-zhang.github.io/BrowseComp-V3/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[36].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.20235"
          },
          {
            "url": "https://github.com/pty12345/ScholarQuest"
          },
          {
            "url": "https://arxiv.org/html/2606.20235"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[36].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.20235"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[36].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.20235"
          },
          {
            "url": "https://github.com/pty12345/ScholarQuest"
          },
          {
            "url": "https://arxiv.org/html/2606.20235"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[36].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.20235"
          },
          {
            "url": "https://github.com/pty12345/ScholarQuest"
          },
          {
            "url": "https://arxiv.org/html/2606.20235"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[36].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.20235"
          },
          {
            "url": "https://github.com/pty12345/ScholarQuest"
          },
          {
            "url": "https://arxiv.org/html/2606.20235"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[36].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.20235"
          },
          {
            "url": "https://github.com/pty12345/ScholarQuest"
          },
          {
            "url": "https://arxiv.org/html/2606.20235"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[36].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.20235"
          },
          {
            "url": "https://github.com/pty12345/ScholarQuest"
          },
          {
            "url": "https://arxiv.org/html/2606.20235"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[36].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.20235"
          },
          {
            "url": "https://github.com/pty12345/ScholarQuest"
          },
          {
            "url": "https://arxiv.org/html/2606.20235"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[36].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.20235"
          },
          {
            "url": "https://github.com/pty12345/ScholarQuest"
          },
          {
            "url": "https://arxiv.org/html/2606.20235"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[36].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.20235"
          },
          {
            "url": "https://github.com/pty12345/ScholarQuest"
          },
          {
            "url": "https://arxiv.org/html/2606.20235"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[36].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.20235"
          },
          {
            "url": "https://github.com/pty12345/ScholarQuest"
          },
          {
            "url": "https://arxiv.org/html/2606.20235"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[37].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.05975"
          },
          {
            "url": "https://arxiv.org/html/2602.05975v2"
          },
          {
            "url": "https://github.com/HughieHu/Sage"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[37].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.05975"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[37].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.05975"
          },
          {
            "url": "https://arxiv.org/html/2602.05975v2"
          },
          {
            "url": "https://github.com/HughieHu/Sage"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[37].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.05975"
          },
          {
            "url": "https://arxiv.org/html/2602.05975v2"
          },
          {
            "url": "https://github.com/HughieHu/Sage"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[37].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.05975"
          },
          {
            "url": "https://arxiv.org/html/2602.05975v2"
          },
          {
            "url": "https://github.com/HughieHu/Sage"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[37].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.05975"
          },
          {
            "url": "https://arxiv.org/html/2602.05975v2"
          },
          {
            "url": "https://github.com/HughieHu/Sage"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[37].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.05975"
          },
          {
            "url": "https://arxiv.org/html/2602.05975v2"
          },
          {
            "url": "https://github.com/HughieHu/Sage"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[37].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.05975"
          },
          {
            "url": "https://arxiv.org/html/2602.05975v2"
          },
          {
            "url": "https://github.com/HughieHu/Sage"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[37].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.05975"
          },
          {
            "url": "https://arxiv.org/html/2602.05975v2"
          },
          {
            "url": "https://github.com/HughieHu/Sage"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[37].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.05975"
          },
          {
            "url": "https://arxiv.org/html/2602.05975v2"
          },
          {
            "url": "https://github.com/HughieHu/Sage"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[37].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.05975"
          },
          {
            "url": "https://arxiv.org/html/2602.05975v2"
          },
          {
            "url": "https://github.com/HughieHu/Sage"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[38].name",
        "citations": [
          {
            "url": "https://allenai.org/asta/bench"
          },
          {
            "url": "https://github.com/allenai/asta-bench"
          },
          {
            "url": "https://arxiv.org/abs/2510.21652"
          },
          {
            "url": "https://allenai.org/blog/astabench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[38].canonical_url",
        "citations": [
          {
            "url": "https://allenai.org/asta/bench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[38].category",
        "citations": [
          {
            "url": "https://allenai.org/asta/bench"
          },
          {
            "url": "https://github.com/allenai/asta-bench"
          },
          {
            "url": "https://arxiv.org/abs/2510.21652"
          },
          {
            "url": "https://allenai.org/blog/astabench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[38].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://allenai.org/asta/bench"
          },
          {
            "url": "https://github.com/allenai/asta-bench"
          },
          {
            "url": "https://arxiv.org/abs/2510.21652"
          },
          {
            "url": "https://allenai.org/blog/astabench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[38].measures",
        "citations": [
          {
            "url": "https://allenai.org/asta/bench"
          },
          {
            "url": "https://github.com/allenai/asta-bench"
          },
          {
            "url": "https://arxiv.org/abs/2510.21652"
          },
          {
            "url": "https://allenai.org/blog/astabench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[38].task_data",
        "citations": [
          {
            "url": "https://allenai.org/asta/bench"
          },
          {
            "url": "https://github.com/allenai/asta-bench"
          },
          {
            "url": "https://arxiv.org/abs/2510.21652"
          },
          {
            "url": "https://allenai.org/blog/astabench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[38].evaluation_code",
        "citations": [
          {
            "url": "https://allenai.org/asta/bench"
          },
          {
            "url": "https://github.com/allenai/asta-bench"
          },
          {
            "url": "https://arxiv.org/abs/2510.21652"
          },
          {
            "url": "https://allenai.org/blog/astabench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[38].judging_criteria",
        "citations": [
          {
            "url": "https://allenai.org/asta/bench"
          },
          {
            "url": "https://github.com/allenai/asta-bench"
          },
          {
            "url": "https://arxiv.org/abs/2510.21652"
          },
          {
            "url": "https://allenai.org/blog/astabench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[38].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://allenai.org/asta/bench"
          },
          {
            "url": "https://github.com/allenai/asta-bench"
          },
          {
            "url": "https://arxiv.org/abs/2510.21652"
          },
          {
            "url": "https://allenai.org/blog/astabench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[38].reported_results_provenance",
        "citations": [
          {
            "url": "https://allenai.org/asta/bench"
          },
          {
            "url": "https://github.com/allenai/asta-bench"
          },
          {
            "url": "https://arxiv.org/abs/2510.21652"
          },
          {
            "url": "https://allenai.org/blog/astabench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[38].evidence",
        "citations": [
          {
            "url": "https://allenai.org/asta/bench"
          },
          {
            "url": "https://github.com/allenai/asta-bench"
          },
          {
            "url": "https://arxiv.org/abs/2510.21652"
          },
          {
            "url": "https://allenai.org/blog/astabench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[39].name",
        "citations": [
          {
            "url": "https://github.com/microsoft/LiveDRBench"
          },
          {
            "url": "https://arxiv.org/html/2508.04183"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[39].canonical_url",
        "citations": [
          {
            "url": "https://github.com/microsoft/LiveDRBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[39].category",
        "citations": [
          {
            "url": "https://github.com/microsoft/LiveDRBench"
          },
          {
            "url": "https://arxiv.org/html/2508.04183"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[39].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/microsoft/LiveDRBench"
          },
          {
            "url": "https://arxiv.org/html/2508.04183"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[39].measures",
        "citations": [
          {
            "url": "https://github.com/microsoft/LiveDRBench"
          },
          {
            "url": "https://arxiv.org/html/2508.04183"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[39].task_data",
        "citations": [
          {
            "url": "https://github.com/microsoft/LiveDRBench"
          },
          {
            "url": "https://arxiv.org/html/2508.04183"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[39].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/microsoft/LiveDRBench"
          },
          {
            "url": "https://arxiv.org/html/2508.04183"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[39].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/microsoft/LiveDRBench"
          },
          {
            "url": "https://arxiv.org/html/2508.04183"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[39].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/microsoft/LiveDRBench"
          },
          {
            "url": "https://arxiv.org/html/2508.04183"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[39].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/microsoft/LiveDRBench"
          },
          {
            "url": "https://arxiv.org/html/2508.04183"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[39].evidence",
        "citations": [
          {
            "url": "https://github.com/microsoft/LiveDRBench"
          },
          {
            "url": "https://arxiv.org/html/2508.04183"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[40].name",
        "citations": [
          {
            "url": "https://github.com/hanjanghoon/DEER"
          },
          {
            "url": "https://arxiv.org/html/2512.17776v2"
          },
          {
            "url": "https://huggingface.co/datasets/LG-AI-Research/DEER-Deep-Research-Benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[40].canonical_url",
        "citations": [
          {
            "url": "https://github.com/hanjanghoon/DEER"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[40].category",
        "citations": [
          {
            "url": "https://github.com/hanjanghoon/DEER"
          },
          {
            "url": "https://arxiv.org/html/2512.17776v2"
          },
          {
            "url": "https://huggingface.co/datasets/LG-AI-Research/DEER-Deep-Research-Benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[40].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/hanjanghoon/DEER"
          },
          {
            "url": "https://arxiv.org/html/2512.17776v2"
          },
          {
            "url": "https://huggingface.co/datasets/LG-AI-Research/DEER-Deep-Research-Benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[40].measures",
        "citations": [
          {
            "url": "https://github.com/hanjanghoon/DEER"
          },
          {
            "url": "https://arxiv.org/html/2512.17776v2"
          },
          {
            "url": "https://huggingface.co/datasets/LG-AI-Research/DEER-Deep-Research-Benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[40].task_data",
        "citations": [
          {
            "url": "https://github.com/hanjanghoon/DEER"
          },
          {
            "url": "https://arxiv.org/html/2512.17776v2"
          },
          {
            "url": "https://huggingface.co/datasets/LG-AI-Research/DEER-Deep-Research-Benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[40].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/hanjanghoon/DEER"
          },
          {
            "url": "https://arxiv.org/html/2512.17776v2"
          },
          {
            "url": "https://huggingface.co/datasets/LG-AI-Research/DEER-Deep-Research-Benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[40].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/hanjanghoon/DEER"
          },
          {
            "url": "https://arxiv.org/html/2512.17776v2"
          },
          {
            "url": "https://huggingface.co/datasets/LG-AI-Research/DEER-Deep-Research-Benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[40].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/hanjanghoon/DEER"
          },
          {
            "url": "https://arxiv.org/html/2512.17776v2"
          },
          {
            "url": "https://huggingface.co/datasets/LG-AI-Research/DEER-Deep-Research-Benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[40].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/hanjanghoon/DEER"
          },
          {
            "url": "https://arxiv.org/html/2512.17776v2"
          },
          {
            "url": "https://huggingface.co/datasets/LG-AI-Research/DEER-Deep-Research-Benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[40].evidence",
        "citations": [
          {
            "url": "https://github.com/hanjanghoon/DEER"
          },
          {
            "url": "https://arxiv.org/html/2512.17776v2"
          },
          {
            "url": "https://huggingface.co/datasets/LG-AI-Research/DEER-Deep-Research-Benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[41].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.14306"
          },
          {
            "url": "https://github.com/sjtu-sai-agents/PaSaMaster"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[41].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.14306"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[41].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.14306"
          },
          {
            "url": "https://github.com/sjtu-sai-agents/PaSaMaster"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[41].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.14306"
          },
          {
            "url": "https://github.com/sjtu-sai-agents/PaSaMaster"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[41].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.14306"
          },
          {
            "url": "https://github.com/sjtu-sai-agents/PaSaMaster"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[41].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.14306"
          },
          {
            "url": "https://github.com/sjtu-sai-agents/PaSaMaster"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[41].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.14306"
          },
          {
            "url": "https://github.com/sjtu-sai-agents/PaSaMaster"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[41].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.14306"
          },
          {
            "url": "https://github.com/sjtu-sai-agents/PaSaMaster"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[41].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.14306"
          },
          {
            "url": "https://github.com/sjtu-sai-agents/PaSaMaster"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[41].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.14306"
          },
          {
            "url": "https://github.com/sjtu-sai-agents/PaSaMaster"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[41].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2605.14306"
          },
          {
            "url": "https://github.com/sjtu-sai-agents/PaSaMaster"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[42].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.25256"
          },
          {
            "url": "https://github.com/CherYou/AutoResearchBench"
          },
          {
            "url": "https://cheryou.github.io/autoresearchbench.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/Lk123/AutoResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[42].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.25256"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[42].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.25256"
          },
          {
            "url": "https://github.com/CherYou/AutoResearchBench"
          },
          {
            "url": "https://cheryou.github.io/autoresearchbench.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/Lk123/AutoResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[42].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.25256"
          },
          {
            "url": "https://github.com/CherYou/AutoResearchBench"
          },
          {
            "url": "https://cheryou.github.io/autoresearchbench.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/Lk123/AutoResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[42].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.25256"
          },
          {
            "url": "https://github.com/CherYou/AutoResearchBench"
          },
          {
            "url": "https://cheryou.github.io/autoresearchbench.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/Lk123/AutoResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[42].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.25256"
          },
          {
            "url": "https://github.com/CherYou/AutoResearchBench"
          },
          {
            "url": "https://cheryou.github.io/autoresearchbench.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/Lk123/AutoResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[42].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.25256"
          },
          {
            "url": "https://github.com/CherYou/AutoResearchBench"
          },
          {
            "url": "https://cheryou.github.io/autoresearchbench.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/Lk123/AutoResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[42].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.25256"
          },
          {
            "url": "https://github.com/CherYou/AutoResearchBench"
          },
          {
            "url": "https://cheryou.github.io/autoresearchbench.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/Lk123/AutoResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[42].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.25256"
          },
          {
            "url": "https://github.com/CherYou/AutoResearchBench"
          },
          {
            "url": "https://cheryou.github.io/autoresearchbench.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/Lk123/AutoResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[42].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.25256"
          },
          {
            "url": "https://github.com/CherYou/AutoResearchBench"
          },
          {
            "url": "https://cheryou.github.io/autoresearchbench.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/Lk123/AutoResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[42].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.25256"
          },
          {
            "url": "https://github.com/CherYou/AutoResearchBench"
          },
          {
            "url": "https://cheryou.github.io/autoresearchbench.github.io/"
          },
          {
            "url": "https://huggingface.co/datasets/Lk123/AutoResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[43].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.09251"
          },
          {
            "url": "https://research.ibm.com/publications/drbencher-can-your-agent-identify-the-entity-retrieve-its-properties-and-do-the-math"
          },
          {
            "url": "https://arxiv.org/html/2604.09251v3"
          },
          {
            "url": "https://github.com/IBM/DrBencher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[43].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.09251"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[43].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.09251"
          },
          {
            "url": "https://research.ibm.com/publications/drbencher-can-your-agent-identify-the-entity-retrieve-its-properties-and-do-the-math"
          },
          {
            "url": "https://arxiv.org/html/2604.09251v3"
          },
          {
            "url": "https://github.com/IBM/DrBencher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[43].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.09251"
          },
          {
            "url": "https://research.ibm.com/publications/drbencher-can-your-agent-identify-the-entity-retrieve-its-properties-and-do-the-math"
          },
          {
            "url": "https://arxiv.org/html/2604.09251v3"
          },
          {
            "url": "https://github.com/IBM/DrBencher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[43].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.09251"
          },
          {
            "url": "https://research.ibm.com/publications/drbencher-can-your-agent-identify-the-entity-retrieve-its-properties-and-do-the-math"
          },
          {
            "url": "https://arxiv.org/html/2604.09251v3"
          },
          {
            "url": "https://github.com/IBM/DrBencher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[43].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.09251"
          },
          {
            "url": "https://research.ibm.com/publications/drbencher-can-your-agent-identify-the-entity-retrieve-its-properties-and-do-the-math"
          },
          {
            "url": "https://arxiv.org/html/2604.09251v3"
          },
          {
            "url": "https://github.com/IBM/DrBencher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[43].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.09251"
          },
          {
            "url": "https://research.ibm.com/publications/drbencher-can-your-agent-identify-the-entity-retrieve-its-properties-and-do-the-math"
          },
          {
            "url": "https://arxiv.org/html/2604.09251v3"
          },
          {
            "url": "https://github.com/IBM/DrBencher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[43].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.09251"
          },
          {
            "url": "https://research.ibm.com/publications/drbencher-can-your-agent-identify-the-entity-retrieve-its-properties-and-do-the-math"
          },
          {
            "url": "https://arxiv.org/html/2604.09251v3"
          },
          {
            "url": "https://github.com/IBM/DrBencher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[43].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.09251"
          },
          {
            "url": "https://research.ibm.com/publications/drbencher-can-your-agent-identify-the-entity-retrieve-its-properties-and-do-the-math"
          },
          {
            "url": "https://arxiv.org/html/2604.09251v3"
          },
          {
            "url": "https://github.com/IBM/DrBencher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[43].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.09251"
          },
          {
            "url": "https://research.ibm.com/publications/drbencher-can-your-agent-identify-the-entity-retrieve-its-properties-and-do-the-math"
          },
          {
            "url": "https://arxiv.org/html/2604.09251v3"
          },
          {
            "url": "https://github.com/IBM/DrBencher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[43].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.09251"
          },
          {
            "url": "https://research.ibm.com/publications/drbencher-can-your-agent-identify-the-entity-retrieve-its-properties-and-do-the-math"
          },
          {
            "url": "https://arxiv.org/html/2604.09251v3"
          },
          {
            "url": "https://github.com/IBM/DrBencher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[44].name",
        "citations": [
          {
            "url": "https://github.com/reka-ai/research-eval"
          },
          {
            "url": "https://reka.ai/news/introducing-research-eval-a-benchmark-for-search-augmented-llms"
          },
          {
            "url": "https://github.com/ndurner/web-research-eval"
          },
          {
            "url": "https://ndurner.github.io/reka-websearch-benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[44].canonical_url",
        "citations": [
          {
            "url": "https://github.com/reka-ai/research-eval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[44].category",
        "citations": [
          {
            "url": "https://github.com/reka-ai/research-eval"
          },
          {
            "url": "https://reka.ai/news/introducing-research-eval-a-benchmark-for-search-augmented-llms"
          },
          {
            "url": "https://github.com/ndurner/web-research-eval"
          },
          {
            "url": "https://ndurner.github.io/reka-websearch-benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[44].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/reka-ai/research-eval"
          },
          {
            "url": "https://reka.ai/news/introducing-research-eval-a-benchmark-for-search-augmented-llms"
          },
          {
            "url": "https://github.com/ndurner/web-research-eval"
          },
          {
            "url": "https://ndurner.github.io/reka-websearch-benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[44].measures",
        "citations": [
          {
            "url": "https://github.com/reka-ai/research-eval"
          },
          {
            "url": "https://reka.ai/news/introducing-research-eval-a-benchmark-for-search-augmented-llms"
          },
          {
            "url": "https://github.com/ndurner/web-research-eval"
          },
          {
            "url": "https://ndurner.github.io/reka-websearch-benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[44].task_data",
        "citations": [
          {
            "url": "https://github.com/reka-ai/research-eval"
          },
          {
            "url": "https://reka.ai/news/introducing-research-eval-a-benchmark-for-search-augmented-llms"
          },
          {
            "url": "https://github.com/ndurner/web-research-eval"
          },
          {
            "url": "https://ndurner.github.io/reka-websearch-benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[44].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/reka-ai/research-eval"
          },
          {
            "url": "https://reka.ai/news/introducing-research-eval-a-benchmark-for-search-augmented-llms"
          },
          {
            "url": "https://github.com/ndurner/web-research-eval"
          },
          {
            "url": "https://ndurner.github.io/reka-websearch-benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[44].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/reka-ai/research-eval"
          },
          {
            "url": "https://reka.ai/news/introducing-research-eval-a-benchmark-for-search-augmented-llms"
          },
          {
            "url": "https://github.com/ndurner/web-research-eval"
          },
          {
            "url": "https://ndurner.github.io/reka-websearch-benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[44].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/reka-ai/research-eval"
          },
          {
            "url": "https://reka.ai/news/introducing-research-eval-a-benchmark-for-search-augmented-llms"
          },
          {
            "url": "https://github.com/ndurner/web-research-eval"
          },
          {
            "url": "https://ndurner.github.io/reka-websearch-benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[44].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/reka-ai/research-eval"
          },
          {
            "url": "https://reka.ai/news/introducing-research-eval-a-benchmark-for-search-augmented-llms"
          },
          {
            "url": "https://github.com/ndurner/web-research-eval"
          },
          {
            "url": "https://ndurner.github.io/reka-websearch-benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[44].evidence",
        "citations": [
          {
            "url": "https://github.com/reka-ai/research-eval"
          },
          {
            "url": "https://reka.ai/news/introducing-research-eval-a-benchmark-for-search-augmented-llms"
          },
          {
            "url": "https://github.com/ndurner/web-research-eval"
          },
          {
            "url": "https://ndurner.github.io/reka-websearch-benchmark"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[45].name",
        "citations": [
          {
            "url": "https://github.com/VibeBench/VibeSearchBench"
          },
          {
            "url": "https://arxiv.org/html/2605.27882"
          },
          {
            "url": "https://vibebench.github.io/VibeSearchBench.github.io/index.html"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[45].canonical_url",
        "citations": [
          {
            "url": "https://github.com/VibeBench/VibeSearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[45].category",
        "citations": [
          {
            "url": "https://github.com/VibeBench/VibeSearchBench"
          },
          {
            "url": "https://arxiv.org/html/2605.27882"
          },
          {
            "url": "https://vibebench.github.io/VibeSearchBench.github.io/index.html"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[45].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/VibeBench/VibeSearchBench"
          },
          {
            "url": "https://arxiv.org/html/2605.27882"
          },
          {
            "url": "https://vibebench.github.io/VibeSearchBench.github.io/index.html"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[45].measures",
        "citations": [
          {
            "url": "https://github.com/VibeBench/VibeSearchBench"
          },
          {
            "url": "https://arxiv.org/html/2605.27882"
          },
          {
            "url": "https://vibebench.github.io/VibeSearchBench.github.io/index.html"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[45].task_data",
        "citations": [
          {
            "url": "https://github.com/VibeBench/VibeSearchBench"
          },
          {
            "url": "https://arxiv.org/html/2605.27882"
          },
          {
            "url": "https://vibebench.github.io/VibeSearchBench.github.io/index.html"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[45].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/VibeBench/VibeSearchBench"
          },
          {
            "url": "https://arxiv.org/html/2605.27882"
          },
          {
            "url": "https://vibebench.github.io/VibeSearchBench.github.io/index.html"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[45].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/VibeBench/VibeSearchBench"
          },
          {
            "url": "https://arxiv.org/html/2605.27882"
          },
          {
            "url": "https://vibebench.github.io/VibeSearchBench.github.io/index.html"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[45].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/VibeBench/VibeSearchBench"
          },
          {
            "url": "https://arxiv.org/html/2605.27882"
          },
          {
            "url": "https://vibebench.github.io/VibeSearchBench.github.io/index.html"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[45].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/VibeBench/VibeSearchBench"
          },
          {
            "url": "https://arxiv.org/html/2605.27882"
          },
          {
            "url": "https://vibebench.github.io/VibeSearchBench.github.io/index.html"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[45].evidence",
        "citations": [
          {
            "url": "https://github.com/VibeBench/VibeSearchBench"
          },
          {
            "url": "https://arxiv.org/html/2605.27882"
          },
          {
            "url": "https://vibebench.github.io/VibeSearchBench.github.io/index.html"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[46].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.20168"
          },
          {
            "url": "https://arxiv.org/html/2510.20168"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://huggingface.co/datasets/AIDC-AI/DeepWideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[46].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.20168"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[46].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.20168"
          },
          {
            "url": "https://arxiv.org/html/2510.20168"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://huggingface.co/datasets/AIDC-AI/DeepWideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[46].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.20168"
          },
          {
            "url": "https://arxiv.org/html/2510.20168"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://huggingface.co/datasets/AIDC-AI/DeepWideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[46].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.20168"
          },
          {
            "url": "https://arxiv.org/html/2510.20168"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://huggingface.co/datasets/AIDC-AI/DeepWideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[46].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.20168"
          },
          {
            "url": "https://arxiv.org/html/2510.20168"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://huggingface.co/datasets/AIDC-AI/DeepWideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[46].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.20168"
          },
          {
            "url": "https://arxiv.org/html/2510.20168"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://huggingface.co/datasets/AIDC-AI/DeepWideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[46].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.20168"
          },
          {
            "url": "https://arxiv.org/html/2510.20168"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://huggingface.co/datasets/AIDC-AI/DeepWideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[46].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.20168"
          },
          {
            "url": "https://arxiv.org/html/2510.20168"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://huggingface.co/datasets/AIDC-AI/DeepWideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[46].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.20168"
          },
          {
            "url": "https://arxiv.org/html/2510.20168"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://huggingface.co/datasets/AIDC-AI/DeepWideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[46].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.20168"
          },
          {
            "url": "https://arxiv.org/html/2510.20168"
          },
          {
            "url": "https://github.com/AIDC-AI/Marco-Search-Agent"
          },
          {
            "url": "https://huggingface.co/datasets/AIDC-AI/DeepWideSearch"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[47].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.05748"
          },
          {
            "url": "https://arxiv.org/pdf/2508.05748"
          },
          {
            "url": "https://github.com/alibaba-nlp/deepresearch/blob/main/WebAgent/WebWatcher/README.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[47].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.05748"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[47].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.05748"
          },
          {
            "url": "https://arxiv.org/pdf/2508.05748"
          },
          {
            "url": "https://github.com/alibaba-nlp/deepresearch/blob/main/WebAgent/WebWatcher/README.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[47].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.05748"
          },
          {
            "url": "https://arxiv.org/pdf/2508.05748"
          },
          {
            "url": "https://github.com/alibaba-nlp/deepresearch/blob/main/WebAgent/WebWatcher/README.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[47].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.05748"
          },
          {
            "url": "https://arxiv.org/pdf/2508.05748"
          },
          {
            "url": "https://github.com/alibaba-nlp/deepresearch/blob/main/WebAgent/WebWatcher/README.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[47].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.05748"
          },
          {
            "url": "https://arxiv.org/pdf/2508.05748"
          },
          {
            "url": "https://github.com/alibaba-nlp/deepresearch/blob/main/WebAgent/WebWatcher/README.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[47].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.05748"
          },
          {
            "url": "https://arxiv.org/pdf/2508.05748"
          },
          {
            "url": "https://github.com/alibaba-nlp/deepresearch/blob/main/WebAgent/WebWatcher/README.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[47].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.05748"
          },
          {
            "url": "https://arxiv.org/pdf/2508.05748"
          },
          {
            "url": "https://github.com/alibaba-nlp/deepresearch/blob/main/WebAgent/WebWatcher/README.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[47].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.05748"
          },
          {
            "url": "https://arxiv.org/pdf/2508.05748"
          },
          {
            "url": "https://github.com/alibaba-nlp/deepresearch/blob/main/WebAgent/WebWatcher/README.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[47].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.05748"
          },
          {
            "url": "https://arxiv.org/pdf/2508.05748"
          },
          {
            "url": "https://github.com/alibaba-nlp/deepresearch/blob/main/WebAgent/WebWatcher/README.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[47].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.05748"
          },
          {
            "url": "https://arxiv.org/pdf/2508.05748"
          },
          {
            "url": "https://github.com/alibaba-nlp/deepresearch/blob/main/WebAgent/WebWatcher/README.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[48].name",
        "citations": [
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          },
          {
            "url": "https://arxiv.org/html/2507.05495v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[48].canonical_url",
        "citations": [
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[48].category",
        "citations": [
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          },
          {
            "url": "https://arxiv.org/html/2507.05495v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[48].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          },
          {
            "url": "https://arxiv.org/html/2507.05495v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[48].measures",
        "citations": [
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          },
          {
            "url": "https://arxiv.org/html/2507.05495v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[48].task_data",
        "citations": [
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          },
          {
            "url": "https://arxiv.org/html/2507.05495v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[48].evaluation_code",
        "citations": [
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          },
          {
            "url": "https://arxiv.org/html/2507.05495v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[48].judging_criteria",
        "citations": [
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          },
          {
            "url": "https://arxiv.org/html/2507.05495v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[48].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          },
          {
            "url": "https://arxiv.org/html/2507.05495v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[48].reported_results_provenance",
        "citations": [
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          },
          {
            "url": "https://arxiv.org/html/2507.05495v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[48].evidence",
        "citations": [
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          },
          {
            "url": "https://arxiv.org/html/2507.05495v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[49].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.15345"
          },
          {
            "url": "https://arxiv.org/html/2606.15345"
          },
          {
            "url": "https://github.com/paddler2022/XBCP"
          },
          {
            "url": "https://huggingface.co/datasets/UTokyo-Yokoya-Lab/XBCP"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[49].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.15345"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[49].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.15345"
          },
          {
            "url": "https://arxiv.org/html/2606.15345"
          },
          {
            "url": "https://github.com/paddler2022/XBCP"
          },
          {
            "url": "https://huggingface.co/datasets/UTokyo-Yokoya-Lab/XBCP"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[49].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.15345"
          },
          {
            "url": "https://arxiv.org/html/2606.15345"
          },
          {
            "url": "https://github.com/paddler2022/XBCP"
          },
          {
            "url": "https://huggingface.co/datasets/UTokyo-Yokoya-Lab/XBCP"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[49].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.15345"
          },
          {
            "url": "https://arxiv.org/html/2606.15345"
          },
          {
            "url": "https://github.com/paddler2022/XBCP"
          },
          {
            "url": "https://huggingface.co/datasets/UTokyo-Yokoya-Lab/XBCP"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[49].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.15345"
          },
          {
            "url": "https://arxiv.org/html/2606.15345"
          },
          {
            "url": "https://github.com/paddler2022/XBCP"
          },
          {
            "url": "https://huggingface.co/datasets/UTokyo-Yokoya-Lab/XBCP"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[49].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.15345"
          },
          {
            "url": "https://arxiv.org/html/2606.15345"
          },
          {
            "url": "https://github.com/paddler2022/XBCP"
          },
          {
            "url": "https://huggingface.co/datasets/UTokyo-Yokoya-Lab/XBCP"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[49].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.15345"
          },
          {
            "url": "https://arxiv.org/html/2606.15345"
          },
          {
            "url": "https://github.com/paddler2022/XBCP"
          },
          {
            "url": "https://huggingface.co/datasets/UTokyo-Yokoya-Lab/XBCP"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[49].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.15345"
          },
          {
            "url": "https://arxiv.org/html/2606.15345"
          },
          {
            "url": "https://github.com/paddler2022/XBCP"
          },
          {
            "url": "https://huggingface.co/datasets/UTokyo-Yokoya-Lab/XBCP"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[49].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.15345"
          },
          {
            "url": "https://arxiv.org/html/2606.15345"
          },
          {
            "url": "https://github.com/paddler2022/XBCP"
          },
          {
            "url": "https://huggingface.co/datasets/UTokyo-Yokoya-Lab/XBCP"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[49].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.15345"
          },
          {
            "url": "https://arxiv.org/html/2606.15345"
          },
          {
            "url": "https://github.com/paddler2022/XBCP"
          },
          {
            "url": "https://huggingface.co/datasets/UTokyo-Yokoya-Lab/XBCP"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[50].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.02404"
          },
          {
            "url": "https://arxiv.org/html/2606.02404"
          },
          {
            "url": "https://github.com/prometheus-eval/K-BrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/prometheus-eval/k-browsecomp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[50].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.02404"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[50].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.02404"
          },
          {
            "url": "https://arxiv.org/html/2606.02404"
          },
          {
            "url": "https://github.com/prometheus-eval/K-BrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/prometheus-eval/k-browsecomp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[50].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.02404"
          },
          {
            "url": "https://arxiv.org/html/2606.02404"
          },
          {
            "url": "https://github.com/prometheus-eval/K-BrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/prometheus-eval/k-browsecomp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[50].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.02404"
          },
          {
            "url": "https://arxiv.org/html/2606.02404"
          },
          {
            "url": "https://github.com/prometheus-eval/K-BrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/prometheus-eval/k-browsecomp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[50].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.02404"
          },
          {
            "url": "https://arxiv.org/html/2606.02404"
          },
          {
            "url": "https://github.com/prometheus-eval/K-BrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/prometheus-eval/k-browsecomp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[50].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.02404"
          },
          {
            "url": "https://arxiv.org/html/2606.02404"
          },
          {
            "url": "https://github.com/prometheus-eval/K-BrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/prometheus-eval/k-browsecomp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[50].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.02404"
          },
          {
            "url": "https://arxiv.org/html/2606.02404"
          },
          {
            "url": "https://github.com/prometheus-eval/K-BrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/prometheus-eval/k-browsecomp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[50].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.02404"
          },
          {
            "url": "https://arxiv.org/html/2606.02404"
          },
          {
            "url": "https://github.com/prometheus-eval/K-BrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/prometheus-eval/k-browsecomp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[50].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.02404"
          },
          {
            "url": "https://arxiv.org/html/2606.02404"
          },
          {
            "url": "https://github.com/prometheus-eval/K-BrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/prometheus-eval/k-browsecomp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[50].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2606.02404"
          },
          {
            "url": "https://arxiv.org/html/2606.02404"
          },
          {
            "url": "https://github.com/prometheus-eval/K-BrowseComp"
          },
          {
            "url": "https://huggingface.co/datasets/prometheus-eval/k-browsecomp"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[51].name",
        "citations": [
          {
            "url": "https://aclanthology.org/2026.acl-long.1249/"
          },
          {
            "url": "https://arxiv.org/html/2601.10504v1"
          },
          {
            "url": "https://github.com/iNLP-Lab/DR-Arena"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[51].canonical_url",
        "citations": [
          {
            "url": "https://aclanthology.org/2026.acl-long.1249/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[51].category",
        "citations": [
          {
            "url": "https://aclanthology.org/2026.acl-long.1249/"
          },
          {
            "url": "https://arxiv.org/html/2601.10504v1"
          },
          {
            "url": "https://github.com/iNLP-Lab/DR-Arena"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[51].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://aclanthology.org/2026.acl-long.1249/"
          },
          {
            "url": "https://arxiv.org/html/2601.10504v1"
          },
          {
            "url": "https://github.com/iNLP-Lab/DR-Arena"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[51].measures",
        "citations": [
          {
            "url": "https://aclanthology.org/2026.acl-long.1249/"
          },
          {
            "url": "https://arxiv.org/html/2601.10504v1"
          },
          {
            "url": "https://github.com/iNLP-Lab/DR-Arena"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[51].task_data",
        "citations": [
          {
            "url": "https://aclanthology.org/2026.acl-long.1249/"
          },
          {
            "url": "https://arxiv.org/html/2601.10504v1"
          },
          {
            "url": "https://github.com/iNLP-Lab/DR-Arena"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[51].evaluation_code",
        "citations": [
          {
            "url": "https://aclanthology.org/2026.acl-long.1249/"
          },
          {
            "url": "https://arxiv.org/html/2601.10504v1"
          },
          {
            "url": "https://github.com/iNLP-Lab/DR-Arena"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[51].judging_criteria",
        "citations": [
          {
            "url": "https://aclanthology.org/2026.acl-long.1249/"
          },
          {
            "url": "https://arxiv.org/html/2601.10504v1"
          },
          {
            "url": "https://github.com/iNLP-Lab/DR-Arena"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[51].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://aclanthology.org/2026.acl-long.1249/"
          },
          {
            "url": "https://arxiv.org/html/2601.10504v1"
          },
          {
            "url": "https://github.com/iNLP-Lab/DR-Arena"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[51].reported_results_provenance",
        "citations": [
          {
            "url": "https://aclanthology.org/2026.acl-long.1249/"
          },
          {
            "url": "https://arxiv.org/html/2601.10504v1"
          },
          {
            "url": "https://github.com/iNLP-Lab/DR-Arena"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[51].evidence",
        "citations": [
          {
            "url": "https://aclanthology.org/2026.acl-long.1249/"
          },
          {
            "url": "https://arxiv.org/html/2601.10504v1"
          },
          {
            "url": "https://github.com/iNLP-Lab/DR-Arena"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[52].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.25106"
          },
          {
            "url": "https://github.com/OPPO-PersonalAI/PersonalizedDeepResearchBench"
          },
          {
            "url": "https://huggingface.co/datasets/PersonalAILab/PersonalizedDeepResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[52].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.25106"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[52].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.25106"
          },
          {
            "url": "https://github.com/OPPO-PersonalAI/PersonalizedDeepResearchBench"
          },
          {
            "url": "https://huggingface.co/datasets/PersonalAILab/PersonalizedDeepResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[52].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.25106"
          },
          {
            "url": "https://github.com/OPPO-PersonalAI/PersonalizedDeepResearchBench"
          },
          {
            "url": "https://huggingface.co/datasets/PersonalAILab/PersonalizedDeepResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[52].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.25106"
          },
          {
            "url": "https://github.com/OPPO-PersonalAI/PersonalizedDeepResearchBench"
          },
          {
            "url": "https://huggingface.co/datasets/PersonalAILab/PersonalizedDeepResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[52].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.25106"
          },
          {
            "url": "https://github.com/OPPO-PersonalAI/PersonalizedDeepResearchBench"
          },
          {
            "url": "https://huggingface.co/datasets/PersonalAILab/PersonalizedDeepResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[52].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.25106"
          },
          {
            "url": "https://github.com/OPPO-PersonalAI/PersonalizedDeepResearchBench"
          },
          {
            "url": "https://huggingface.co/datasets/PersonalAILab/PersonalizedDeepResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[52].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.25106"
          },
          {
            "url": "https://github.com/OPPO-PersonalAI/PersonalizedDeepResearchBench"
          },
          {
            "url": "https://huggingface.co/datasets/PersonalAILab/PersonalizedDeepResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[52].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.25106"
          },
          {
            "url": "https://github.com/OPPO-PersonalAI/PersonalizedDeepResearchBench"
          },
          {
            "url": "https://huggingface.co/datasets/PersonalAILab/PersonalizedDeepResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[52].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.25106"
          },
          {
            "url": "https://github.com/OPPO-PersonalAI/PersonalizedDeepResearchBench"
          },
          {
            "url": "https://huggingface.co/datasets/PersonalAILab/PersonalizedDeepResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[52].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2509.25106"
          },
          {
            "url": "https://github.com/OPPO-PersonalAI/PersonalizedDeepResearchBench"
          },
          {
            "url": "https://huggingface.co/datasets/PersonalAILab/PersonalizedDeepResearchBench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[53].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.02190"
          },
          {
            "url": "https://openreview.net/forum?id=EYUG4Su6ZU"
          },
          {
            "url": "https://ar5iv.labs.arxiv.org/html/2510.02190"
          },
          {
            "url": "https://github.com/evigbyen/rigorousbench/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[53].canonical_url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.02190"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[53].category",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.02190"
          },
          {
            "url": "https://openreview.net/forum?id=EYUG4Su6ZU"
          },
          {
            "url": "https://ar5iv.labs.arxiv.org/html/2510.02190"
          },
          {
            "url": "https://github.com/evigbyen/rigorousbench/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[53].eligibility_date_and_evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.02190"
          },
          {
            "url": "https://openreview.net/forum?id=EYUG4Su6ZU"
          },
          {
            "url": "https://ar5iv.labs.arxiv.org/html/2510.02190"
          },
          {
            "url": "https://github.com/evigbyen/rigorousbench/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[53].measures",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.02190"
          },
          {
            "url": "https://openreview.net/forum?id=EYUG4Su6ZU"
          },
          {
            "url": "https://ar5iv.labs.arxiv.org/html/2510.02190"
          },
          {
            "url": "https://github.com/evigbyen/rigorousbench/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[53].task_data",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.02190"
          },
          {
            "url": "https://openreview.net/forum?id=EYUG4Su6ZU"
          },
          {
            "url": "https://ar5iv.labs.arxiv.org/html/2510.02190"
          },
          {
            "url": "https://github.com/evigbyen/rigorousbench/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[53].evaluation_code",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.02190"
          },
          {
            "url": "https://openreview.net/forum?id=EYUG4Su6ZU"
          },
          {
            "url": "https://ar5iv.labs.arxiv.org/html/2510.02190"
          },
          {
            "url": "https://github.com/evigbyen/rigorousbench/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[53].judging_criteria",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.02190"
          },
          {
            "url": "https://openreview.net/forum?id=EYUG4Su6ZU"
          },
          {
            "url": "https://ar5iv.labs.arxiv.org/html/2510.02190"
          },
          {
            "url": "https://github.com/evigbyen/rigorousbench/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[53].reproducibility_and_barriers",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.02190"
          },
          {
            "url": "https://openreview.net/forum?id=EYUG4Su6ZU"
          },
          {
            "url": "https://ar5iv.labs.arxiv.org/html/2510.02190"
          },
          {
            "url": "https://github.com/evigbyen/rigorousbench/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[53].reported_results_provenance",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.02190"
          },
          {
            "url": "https://openreview.net/forum?id=EYUG4Su6ZU"
          },
          {
            "url": "https://ar5iv.labs.arxiv.org/html/2510.02190"
          },
          {
            "url": "https://github.com/evigbyen/rigorousbench/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.benchmarks[53].evidence",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2510.02190"
          },
          {
            "url": "https://openreview.net/forum?id=EYUG4Su6ZU"
          },
          {
            "url": "https://ar5iv.labs.arxiv.org/html/2510.02190"
          },
          {
            "url": "https://github.com/evigbyen/rigorousbench/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[0].name",
        "citations": [
          {
            "url": "https://github.com/tavily-ai/tavily-search-evals"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[0].url",
        "citations": [
          {
            "url": "https://github.com/tavily-ai/tavily-search-evals"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[0].reason",
        "citations": [
          {
            "url": "https://github.com/tavily-ai/tavily-search-evals"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[1].name",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01952"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[1].url",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01952"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[1].reason",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2506.01952"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[2].name",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2504.08942"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[2].url",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2504.08942"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[2].reason",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2504.08942"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[3].name",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2506.02865"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[3].url",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2506.02865"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[3].reason",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2506.02865"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[4].name",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2510.02418"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[4].url",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2510.02418"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[4].reason",
        "citations": [
          {
            "url": "https://doi.org/10.48550/arxiv.2510.02418"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[5].name",
        "citations": [
          {
            "url": "https://github.com/perplexityai/search_evals"
          },
          {
            "url": "https://research.perplexity.ai/articles/architecting-and-evaluating-an-ai-first-search-api"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[5].url",
        "citations": [
          {
            "url": "https://github.com/perplexityai/search_evals"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[5].reason",
        "citations": [
          {
            "url": "https://github.com/perplexityai/search_evals"
          },
          {
            "url": "https://research.perplexity.ai/articles/architecting-and-evaluating-an-ai-first-search-api"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[6].name",
        "citations": [
          {
            "url": "https://github.com/parallel-web/parallel-llms-txt/blob/f6b31ffe/public/blog/deepsearch-qa.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[6].url",
        "citations": [
          {
            "url": "https://github.com/parallel-web/parallel-llms-txt/blob/f6b31ffe/public/blog/deepsearch-qa.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[6].reason",
        "citations": [
          {
            "url": "https://github.com/parallel-web/parallel-llms-txt/blob/f6b31ffe/public/blog/deepsearch-qa.md"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[7].name",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-grounding"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[7].url",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-grounding"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[7].reason",
        "citations": [
          {
            "url": "https://www.kaggle.com/benchmarks/google/facts-grounding"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[8].name",
        "citations": [
          {
            "url": "https://github.com/OSU-NLP-Group/Online-Mind2Web"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[8].url",
        "citations": [
          {
            "url": "https://github.com/OSU-NLP-Group/Online-Mind2Web"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[8].reason",
        "citations": [
          {
            "url": "https://github.com/OSU-NLP-Group/Online-Mind2Web"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[9].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2409.12941"
          },
          {
            "url": "https://github.com/sahil350/frames-eval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[9].url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2409.12941"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[9].reason",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2409.12941"
          },
          {
            "url": "https://github.com/sahil350/frames-eval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[10].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14558"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[10].url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14558"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[10].reason",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2505.14558"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[11].name",
        "citations": [
          {
            "url": "https://github.com/Talc-AI/search-bench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[11].url",
        "citations": [
          {
            "url": "https://github.com/Talc-AI/search-bench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[11].reason",
        "citations": [
          {
            "url": "https://github.com/Talc-AI/search-bench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[12].name",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/DeepResearch/tree/main/WebAgent/WebWatcher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[12].url",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/DeepResearch/tree/main/WebAgent/WebWatcher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[12].reason",
        "citations": [
          {
            "url": "https://github.com/Alibaba-NLP/DeepResearch/tree/main/WebAgent/WebWatcher"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[13].name",
        "citations": [
          {
            "url": "https://github.com/google-deepmind/webquest"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[13].url",
        "citations": [
          {
            "url": "https://github.com/google-deepmind/webquest"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[13].reason",
        "citations": [
          {
            "url": "https://github.com/google-deepmind/webquest"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[14].name",
        "citations": [
          {
            "url": "https://github.com/ServiceNow/drbench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[14].url",
        "citations": [
          {
            "url": "https://github.com/ServiceNow/drbench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[14].reason",
        "citations": [
          {
            "url": "https://github.com/ServiceNow/drbench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[15].name",
        "citations": [
          {
            "url": "https://github.com/aiming-lab/AutoResearchClaw/tree/main/experiments/arc_bench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[15].url",
        "citations": [
          {
            "url": "https://github.com/aiming-lab/AutoResearchClaw/tree/main/experiments/arc_bench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[15].reason",
        "citations": [
          {
            "url": "https://github.com/aiming-lab/AutoResearchClaw/tree/main/experiments/arc_bench"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[16].name",
        "citations": [
          {
            "url": "https://entityenricher.ai/docs/platform/benchmarks"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[16].url",
        "citations": [
          {
            "url": "https://entityenricher.ai/docs/platform/benchmarks"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[16].reason",
        "citations": [
          {
            "url": "https://entityenricher.ai/docs/platform/benchmarks"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[17].name",
        "citations": [
          {
            "url": "https://exa.ai/blog/websets-evals"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[17].url",
        "citations": [
          {
            "url": "https://exa.ai/blog/websets-evals"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[17].reason",
        "citations": [
          {
            "url": "https://exa.ai/blog/websets-evals"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[18].name",
        "citations": [
          {
            "url": "https://parallel.ai/products/findall"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[18].url",
        "citations": [
          {
            "url": "https://parallel.ai/products/findall"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[18].reason",
        "citations": [
          {
            "url": "https://parallel.ai/products/findall"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[19].name",
        "citations": [
          {
            "url": "https://github.com/reka-ai/reka-vibe-eval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[19].url",
        "citations": [
          {
            "url": "https://github.com/reka-ai/reka-vibe-eval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[19].reason",
        "citations": [
          {
            "url": "https://github.com/reka-ai/reka-vibe-eval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[20].name",
        "citations": [
          {
            "url": "https://www.vals.ai/benchmarks/web_search"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[20].url",
        "citations": [
          {
            "url": "https://www.vals.ai/benchmarks/web_search"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[20].reason",
        "citations": [
          {
            "url": "https://www.vals.ai/benchmarks/web_search"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[21].name",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/jxg25/EntiWeave"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[21].url",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/jxg25/EntiWeave"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[21].reason",
        "citations": [
          {
            "url": "https://huggingface.co/datasets/jxg25/EntiWeave"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[22].name",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.findings-acl.988/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[22].url",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.findings-acl.988/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[22].reason",
        "citations": [
          {
            "url": "https://aclanthology.org/2025.findings-acl.988/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[23].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.05334"
          },
          {
            "url": "https://github.com/lmarena/search-arena"
          },
          {
            "url": "https://huggingface.co/datasets/lmarena-ai/search-arena-v1-7k"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[23].url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.05334"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[23].reason",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2506.05334"
          },
          {
            "url": "https://github.com/lmarena/search-arena"
          },
          {
            "url": "https://huggingface.co/datasets/lmarena-ai/search-arena-v1-7k"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[24].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.14683"
          },
          {
            "url": "https://github.com/NJU-LINK/DR3-Eval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[24].url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.14683"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[24].reason",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2604.14683"
          },
          {
            "url": "https://github.com/NJU-LINK/DR3-Eval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[25].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.15019"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[25].url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.15019"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[25].reason",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2602.15019"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[26].name",
        "citations": [
          {
            "url": "https://ai.meta.com/research/publications/gaia-a-benchmark-for-general-ai-assistants/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[26].url",
        "citations": [
          {
            "url": "https://ai.meta.com/research/publications/gaia-a-benchmark-for-general-ai-assistants/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[26].reason",
        "citations": [
          {
            "url": "https://ai.meta.com/research/publications/gaia-a-benchmark-for-general-ai-assistants/"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[27].name",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2501.14249"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[27].url",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2501.14249"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[27].reason",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2501.14249"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[28].name",
        "citations": [
          {
            "url": "https://arxiv.org/html/2509.07968v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[28].url",
        "citations": [
          {
            "url": "https://arxiv.org/html/2509.07968v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[28].reason",
        "citations": [
          {
            "url": "https://arxiv.org/html/2509.07968v1"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[29].name",
        "citations": [
          {
            "url": "https://github.com/yale-nlp/SciArena"
          },
          {
            "url": "https://huggingface.co/datasets/yale-nlp/SciArena"
          },
          {
            "url": "https://huggingface.co/datasets/yale-nlp/SciArena-with-paperbank"
          },
          {
            "url": "https://proceedings.neurips.cc/paper_files/paper/2025/file/9811fe727c94e7ff79f701c94ed1d938-Paper-Datasets_and_Benchmarks_Track.pdf"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[29].url",
        "citations": [
          {
            "url": "https://github.com/yale-nlp/SciArena"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.adjacent_or_uncertain[29].reason",
        "citations": [
          {
            "url": "https://github.com/yale-nlp/SciArena"
          },
          {
            "url": "https://huggingface.co/datasets/yale-nlp/SciArena"
          },
          {
            "url": "https://huggingface.co/datasets/yale-nlp/SciArena-with-paperbank"
          },
          {
            "url": "https://proceedings.neurips.cc/paper_files/paper/2025/file/9811fe727c94e7ff79f701c94ed1d938-Paper-Datasets_and_Benchmarks_Track.pdf"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.summary",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/reka-ai/research-eval"
          }
        ],
        "confidence": "medium"
      },
      {
        "field": "structured.coverage_gaps",
        "citations": [
          {
            "url": "https://arxiv.org/abs/2508.07999"
          },
          {
            "url": "https://github.com/Ayanami0730/deep_research_bench"
          },
          {
            "url": "https://github.com/texttron/BrowseComp-Plus"
          },
          {
            "url": "https://openbenchmarks.com/multi-turn-company-search"
          },
          {
            "url": "https://github.com/reka-ai/research-eval"
          },
          {
            "url": "https://arxiv.org/abs/2606.27595"
          },
          {
            "url": "https://github.com/PALIN2018/BrowseComp-ZH"
          },
          {
            "url": "https://arxiv.org/abs/2608.14747"
          },
          {
            "url": "https://arxiv.org/abs/2602.05975"
          },
          {
            "url": "https://github.com/cxcscmu/Deep-Research-Comparator"
          },
          {
            "url": "https://github.com/yale-nlp/SciArena"
          },
          {
            "url": "https://huggingface.co/datasets/yale-nlp/SciArena"
          },
          {
            "url": "https://huggingface.co/datasets/yale-nlp/SciArena-with-paperbank"
          },
          {
            "url": "https://proceedings.neurips.cc/paper_files/paper/2025/file/9811fe727c94e7ff79f701c94ed1d938-Paper-Datasets_and_Benchmarks_Track.pdf"
          },
          {
            "url": "https://arxiv.org/abs/2506.05334"
          },
          {
            "url": "https://github.com/lmarena/search-arena"
          },
          {
            "url": "https://huggingface.co/datasets/lmarena-ai/search-arena-v1-7k"
          }
        ],
        "confidence": "medium"
      }
    ]
  },
  "usage": {
    "agentComputeUnits": 111.968,
    "searches": 92,
    "emails": 0,
    "phoneNumbers": 0
  },
  "costDollars": {
    "total": 11.6589,
    "agentCompute": 11.1968,
    "search": 0.462,
    "emails": 0,
    "phoneNumbers": 0
  }
}
