Repository object · research-note

Evidence Ledger V1

Accepted research note in the public catalog.

Source path
research/how-we-know/agent-citation-lineage/evidence-ledger-v1.json
Media type
application/json
Object ID
em:research-note:sha256:b7cd747cbe546491dbc2b835cb2df419c09f707d5ad4a94d027a1ebe803d5333
Content digest
29034edff99ae40f03d244998ee1d238917ac387422dea221b80897c94d556b3

Source content

{

"candidate_warrants": [

{

"canonical_proposition": "In one five-system dermatology-review audit, reference identifiability and metadata correctness did not establish sentence-level claim-citation concordance.",

"data_root": "data:keplinger-three-run-five-system-audit",

"derivation_root": "derivation:keplinger-published-tables-and-figure",

"independent_review_status": "pending",

"method_root": "method:human-reference-and-sentence-concordance-audit",

"review_status": "candidate-supported-with-scope",

"warrant_id": "warrant:keplinger-metadata-versus-support",

"work_ids": [

"work:keplinger-dermatology-audit",

"work:keplinger-supplement"

]

},

{

"canonical_proposition": "DeepResearch Bench evaluated binary support for deduplicated statement-URL pairs; numeric values are edition-specific.",

"data_root": "data:deepresearch-bench-100-tasks-and-system-outputs",

"derivation_root": "derivation:deepresearch-bench-edition-specific-tables",

"independent_review_status": "pending",

"method_root": "method:fact-statement-url-support-evaluator",

"review_status": "candidate-supported-with-edition-boundary",

"warrant_id": "warrant:deepresearch-bench-fact",

"work_ids": [

"work:deepresearch-bench-paper",

"work:deepresearch-bench-repository"

]

},

{

"canonical_proposition": "DeepTRACE measured statement-source support and found between-system variation; its Gemini value is internally inconsistent between table and prose.",

"data_root": "data:deeptrace-303-questions-2727-outputs",

"derivation_root": "derivation:deeptrace-tables-and-prose",

"independent_review_status": "pending",

"method_root": "method:deeptrace-statement-source-matrix",

"review_status": "candidate-supported-except-disputed-gemini-number",

"warrant_id": "warrant:deeptrace-support-variation",

"work_ids": [

"work:deeptrace"

]

},

{

"canonical_proposition": "Cited but Not Verified operationalized link access, topical relevance, and factual support separately and reported materially different rates.",

"data_root": "data:cnv-130-drbench-and-browsecomp-queries",

"derivation_root": "derivation:cnv-main-table",

"independent_review_status": "pending",

"method_root": "method:cnv-ast-access-relevance-fact-check",

"review_status": "candidate-supported-with-upstream-dependence",

"warrant_id": "warrant:cnv-link-relevance-support-gap",

"work_ids": [

"work:cited-not-verified"

]

},

{

"canonical_proposition": "Within the Cited but Not Verified harness, the 2-to-150-call ablation reduced reported Fact Check scores for two setups without establishing a general causal law about search depth.",

"data_root": "data:cnv-depth-ablation-two-model-setups",

"derivation_root": "derivation:cnv-depth-ablation-tables",

"independent_review_status": "pending",

"method_root": "method:cnv-ast-access-relevance-fact-check",

"review_status": "candidate-supported-as-within-harness-association",

"warrant_id": "warrant:cnv-search-depth-ablation",

"work_ids": [

"work:cited-not-verified"

]

},

{

"canonical_proposition": "The URL-health study measured HTTP resolution and Wayback presence, not semantic claim support, on outputs including reused DeepResearch Bench material.",

"data_root": "data:url-health-drbench-and-expertqa-urls",

"derivation_root": "derivation:url-health-observational-results",

"independent_review_status": "pending",

"method_root": "method:http-resolution-wayback-classification",

"review_status": "candidate-supported-with-semantic-exclusion",

"warrant_id": "warrant:url-health-resolution",

"work_ids": [

"work:url-health"

]

},

{

"canonical_proposition": "ReportBench compared cited statements with retrieved cited-page content and reported sub-100-percent semantic match rates in its bounded survey-task setting.",

"data_root": "data:reportbench-100-survey-tasks",

"derivation_root": "derivation:reportbench-table-1",

"independent_review_status": "pending",

"method_root": "method:reportbench-statement-source-semantic-match",

"review_status": "candidate-supported-with-judge-limitation",

"warrant_id": "warrant:reportbench-match-rate",

"work_ids": [

"work:reportbench"

]

}

],

"citations": [

{

"citation_occurrence_id": "V2-SOL-01:s1_keplinger",

"correction_ids": [],

"edition_id": "edition:keplinger-vor-2025",

"license": "CC BY-NC 4.0",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"raw_source_id": "s1_keplinger",

"raw_title": "Assessment of Deep Research for dermatology literature reviews: Deep concern over the hype",

"readback": {

"captured_bytes": 111993,

"captured_sha256": "abbaa9f22035230d13152a68aaeecd8e83e40867c9dae4acd1d547e36c866793",

"edition_id": "edition:keplinger-vor-2025",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"resolved_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"retrieval_status": "retrieved"

},

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"resolution_status": "unresolved",

"run_id": "V2-SOL-01",

"source_work_id": "work:keplinger-dermatology-audit",

"span_occurrence_ids": [

"V2-SOL-01:s1_keplinger:sp1a",

"V2-SOL-01:s1_keplinger:sp1b",

"V2-SOL-01:s1_keplinger:sp1c"

]

},

{

"citation_occurrence_id": "V2-SOL-01:s1a_keplinger_data",

"correction_ids": [

"correction:mendeley-file-readback"

],

"edition_id": "edition:keplinger-supplement-v2",

"license": "CC BY 4.0",

"license_treatment": "metadata and quote-minimal landing-page spans; file API required authentication and was not used",

"raw_source_id": "s1a_keplinger_data",

"raw_title": "Supplementary materials of the article: Assessment of Deep Research for Dermatology Literature Reviews: Deep Concern Over the Hype",

"readback": {

"captured_bytes": 116951,

"captured_sha256": "0a2c4c6ed54dd7e2e0ca6cc07aa19b18fa1439e506c2310db778094282d2cebc",

"edition_id": "edition:keplinger-supplement-v2",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://data.mendeley.com/datasets/3s73z9zf3c/2",

"resolved_url": "https://data.mendeley.com/datasets/3s73z9zf3c/2",

"retrieval_status": "retrieved"

},

"requested_url": "https://data.mendeley.com/datasets/3s73z9zf3c/2",

"resolution_status": "unresolved",

"run_id": "V2-SOL-01",

"source_work_id": "work:keplinger-supplement",

"span_occurrence_ids": [

"V2-SOL-01:s1a_keplinger_data:sp1d"

]

},

{

"citation_occurrence_id": "V2-SOL-01:s2_drbench",

"correction_ids": [],

"edition_id": "edition:drbench-iclr-2026",

"license": "unknown",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"raw_source_id": "s2_drbench",

"raw_title": "DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents",

"readback": {

"captured_bytes": 2665857,

"captured_sha256": "8f80ce247f7cc355bb6773f36037e01cc3ba2e2082c77381eaadf0d2b92b021c",

"edition_id": "edition:drbench-iclr-2026",

"http_status": 200,

"media_type": "application/pdf",

"redirect_chain": [],

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/465f22be10e07b301c6ed58f0472f704-Paper-Conference.pdf",

"resolved_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/465f22be10e07b301c6ed58f0472f704-Paper-Conference.pdf",

"retrieval_status": "retrieved"

},

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/465f22be10e07b301c6ed58f0472f704-Paper-Conference.pdf",

"resolution_status": "unresolved",

"run_id": "V2-SOL-01",

"source_work_id": "work:deepresearch-bench-paper",

"span_occurrence_ids": [

"V2-SOL-01:s2_drbench:sp2a",

"V2-SOL-01:s2_drbench:sp2b",

"V2-SOL-01:s2_drbench:sp2c",

"V2-SOL-01:s2_drbench:sp2d"

]

},

{

"citation_occurrence_id": "V2-SOL-01:s2a_drbench_repo",

"correction_ids": [],

"edition_id": "edition:drbench-repo-main-469cce5",

"license": "Apache-2.0",

"license_treatment": "metadata and quote-minimal README spans",

"raw_source_id": "s2a_drbench_repo",

"raw_title": "Ayanami0730/deep_research_bench",

"readback": {

"captured_bytes": 373533,

"captured_sha256": "ec9a4efdb736c6b5ec198cfbc5534eb67b5d421efa12cf795d7990be6c1e3b54",

"edition_id": "edition:drbench-repo-main-469cce5",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://github.com/Ayanami0730/deep_research_bench",

"resolved_url": "https://github.com/Ayanami0730/deep_research_bench",

"retrieval_status": "retrieved"

},

"requested_url": "https://github.com/Ayanami0730/deep_research_bench",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-01",

"source_work_id": "work:deepresearch-bench-repository",

"span_occurrence_ids": [

"V2-SOL-01:s2a_drbench_repo:sp2e"

]

},

{

"citation_occurrence_id": "V2-SOL-01:s3_deeptrace",

"correction_ids": [],

"edition_id": "edition:deeptrace-iclr-2026",

"license": "unknown",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"raw_source_id": "s3_deeptrace",

"raw_title": "DeepTRACE: Auditing Deep Research AI Systems for Tracking Reliability Across Citations and Evidence",

"readback": {

"captured_bytes": 2982839,

"captured_sha256": "dea4981c1066d0240a005f603b3b14419c7e32beb5191b3df847ceb26af3d6b6",

"edition_id": "edition:deeptrace-iclr-2026",

"http_status": 200,

"media_type": "application/pdf",

"redirect_chain": [],

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"resolved_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"retrieval_status": "retrieved"

},

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"resolution_status": "unresolved",

"run_id": "V2-SOL-01",

"source_work_id": "work:deeptrace",

"span_occurrence_ids": [

"V2-SOL-01:s3_deeptrace:sp3a",

"V2-SOL-01:s3_deeptrace:sp3b",

"V2-SOL-01:s3_deeptrace:sp3c",

"V2-SOL-01:s3_deeptrace:sp3d"

]

},

{

"citation_occurrence_id": "V2-SOL-01:s4_cited_not_verified",

"correction_ids": [],

"edition_id": "edition:cnv-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "s4_cited_not_verified",

"raw_title": "Cited but Not Verified: Parsing and Evaluating Source Attribution in LLM Deep Research Agents",

"readback": {

"captured_bytes": 43848,

"captured_sha256": "d1b6d476b3e81460dbc8821757775a8fa589e0579b77167ef0f33d35cb819cb5",

"edition_id": "edition:cnv-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/abs/2605.06635",

"resolved_url": "https://arxiv.org/abs/2605.06635",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/abs/2605.06635",

"resolution_status": "unresolved",

"run_id": "V2-SOL-01",

"source_work_id": "work:cited-not-verified",

"span_occurrence_ids": [

"V2-SOL-01:s4_cited_not_verified:sp4a",

"V2-SOL-01:s4_cited_not_verified:sp4b",

"V2-SOL-01:s4_cited_not_verified:sp4c"

]

},

{

"citation_occurrence_id": "V2-SOL-01:s5_url_health",

"correction_ids": [],

"edition_id": "edition:url-health-arxiv-v1",

"license": "CC0 1.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "s5_url_health",

"raw_title": "Detecting and Correcting Reference Hallucinations in Commercial LLMs and Deep Research Agents",

"readback": {

"captured_bytes": 42408,

"captured_sha256": "3976893e82f91cc9e7826c0d2d08c6caab40778134dbab09ebbfe1f6f63f5397",

"edition_id": "edition:url-health-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/abs/2604.03173",

"resolved_url": "https://arxiv.org/abs/2604.03173",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/abs/2604.03173",

"resolution_status": "unresolved",

"run_id": "V2-SOL-01",

"source_work_id": "work:url-health",

"span_occurrence_ids": [

"V2-SOL-01:s5_url_health:sp5a",

"V2-SOL-01:s5_url_health:sp5b",

"V2-SOL-01:s5_url_health:sp5c"

]

},

{

"citation_occurrence_id": "V2-SOL-02:S1",

"correction_ids": [],

"edition_id": "edition:url-health-arxiv-v1",

"license": "CC0 1.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S1",

"raw_title": "Detecting and Correcting Reference Hallucinations in Commercial LLMs and Deep Research Agents",

"readback": {

"captured_bytes": 302498,

"captured_sha256": "332e5b5cb4b0ee7065b1bbc30436dfdfeafca8130fb87e32b0020026d8bbafe1",

"edition_id": "edition:url-health-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2604.03173v1",

"resolved_url": "https://arxiv.org/html/2604.03173v1",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2604.03173v1",

"resolution_status": "unresolved",

"run_id": "V2-SOL-02",

"source_work_id": "work:url-health",

"span_occurrence_ids": [

"V2-SOL-02:S1:S1_SPAN_1",

"V2-SOL-02:S1:S1_SPAN_2"

]

},

{

"citation_occurrence_id": "V2-SOL-02:S2",

"correction_ids": [],

"edition_id": "edition:keplinger-vor-2025",

"license": "CC BY-NC 4.0",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"raw_source_id": "S2",

"raw_title": "Assessment of Deep Research for dermatology literature reviews: Deep concern over the hype",

"readback": {

"captured_bytes": 111993,

"captured_sha256": "abbaa9f22035230d13152a68aaeecd8e83e40867c9dae4acd1d547e36c866793",

"edition_id": "edition:keplinger-vor-2025",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"resolved_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"retrieval_status": "retrieved"

},

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-02",

"source_work_id": "work:keplinger-dermatology-audit",

"span_occurrence_ids": [

"V2-SOL-02:S2:S2_SPAN_1"

]

},

{

"citation_occurrence_id": "V2-SOL-02:S3",

"correction_ids": [

"correction:mendeley-file-readback"

],

"edition_id": "edition:keplinger-supplement-v2",

"license": "CC BY 4.0",

"license_treatment": "metadata and quote-minimal landing-page spans; file API required authentication and was not used",

"raw_source_id": "S3",

"raw_title": "Supplementary materials of the article: Assessment of Deep Research for Dermatology Literature Reviews: Deep Concern Over the Hype",

"readback": {

"captured_bytes": 116951,

"captured_sha256": "0a2c4c6ed54dd7e2e0ca6cc07aa19b18fa1439e506c2310db778094282d2cebc",

"edition_id": "edition:keplinger-supplement-v2",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://data.mendeley.com/datasets/3s73z9zf3c/2",

"resolved_url": "https://data.mendeley.com/datasets/3s73z9zf3c/2",

"retrieval_status": "retrieved"

},

"requested_url": "https://data.mendeley.com/datasets/3s73z9zf3c/2",

"resolution_status": "unresolved",

"run_id": "V2-SOL-02",

"source_work_id": "work:keplinger-supplement",

"span_occurrence_ids": [

"V2-SOL-02:S3:S3_SPAN_1",

"V2-SOL-02:S3:S3_SPAN_2"

]

},

{

"citation_occurrence_id": "V2-SOL-02:S4",

"correction_ids": [],

"edition_id": "edition:drbench-iclr-2026",

"license": "unknown",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"raw_source_id": "S4",

"raw_title": "DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents",

"readback": {

"captured_bytes": 12692,

"captured_sha256": "a2d811e0d24af89d9bd4b19cc9e8fbc474c67ef3827648550a6560eca2a5e332",

"edition_id": "edition:drbench-iclr-2026",

"http_status": 403,

"media_type": "text/html",

"redirect_chain": [],

"requested_url": "https://openreview.net/pdf?id=hQ0K2Hhq7H",

"resolved_url": "https://openreview.net/pdf?id=hQ0K2Hhq7H",

"retrieval_status": "inaccessible"

},

"requested_url": "https://openreview.net/pdf?id=hQ0K2Hhq7H",

"resolution_status": "unresolved",

"run_id": "V2-SOL-02",

"source_work_id": "work:deepresearch-bench-paper",

"span_occurrence_ids": [

"V2-SOL-02:S4:S4_SPAN_1",

"V2-SOL-02:S4:S4_SPAN_2"

]

},

{

"citation_occurrence_id": "V2-SOL-02:S5",

"correction_ids": [],

"edition_id": "edition:reportbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S5",

"raw_title": "ReportBench: Evaluating Deep Research Agents via Academic Survey Tasks",

"readback": {

"captured_bytes": 157386,

"captured_sha256": "055e568189a402dd510c7c84be60f59465d9a001808936bc25cb0026ac25d267",

"edition_id": "edition:reportbench-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2508.15804v1",

"resolved_url": "https://arxiv.org/html/2508.15804v1",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2508.15804v1",

"resolution_status": "unresolved",

"run_id": "V2-SOL-02",

"source_work_id": "work:reportbench",

"span_occurrence_ids": [

"V2-SOL-02:S5:S5_SPAN_1",

"V2-SOL-02:S5:S5_SPAN_2"

]

},

{

"citation_occurrence_id": "V2-SOL-02:S6",

"correction_ids": [],

"edition_id": "edition:deeptrace-iclr-2026",

"license": "unknown",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"raw_source_id": "S6",

"raw_title": "DeepTRACE: Auditing Deep Research AI Systems for Tracking Reliability Across Citations and Evidence",

"readback": {

"captured_bytes": 10137,

"captured_sha256": "5457daae5b00c1c4105a9c80f5eb39a4027d79a70702071cc2892a00eb27e05d",

"edition_id": "edition:deeptrace-iclr-2026",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/ad08767706825033b99122332293033d-Abstract-Conference.html",

"resolved_url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/ad08767706825033b99122332293033d-Abstract-Conference.html",

"retrieval_status": "retrieved"

},

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/ad08767706825033b99122332293033d-Abstract-Conference.html",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-02",

"source_work_id": "work:deeptrace",

"span_occurrence_ids": [

"V2-SOL-02:S6:S6_SPAN_1",

"V2-SOL-02:S6:S6_SPAN_2"

]

},

{

"citation_occurrence_id": "V2-SOL-02:S7",

"correction_ids": [],

"edition_id": "edition:cnv-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S7",

"raw_title": "Cited but Not Verified: Parsing and Evaluating Source Attribution in LLM Deep Research Agents",

"readback": {

"captured_bytes": 138385,

"captured_sha256": "7c5e3c33f3122b07d176e6679cd6762a95a6babaf4a353686f019f905b5526ca",

"edition_id": "edition:cnv-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2605.06635v1",

"resolved_url": "https://arxiv.org/html/2605.06635v1",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2605.06635v1",

"resolution_status": "unresolved",

"run_id": "V2-SOL-02",

"source_work_id": "work:cited-not-verified",

"span_occurrence_ids": [

"V2-SOL-02:S7:S7_SPAN_1",

"V2-SOL-02:S7:S7_SPAN_2"

]

},

{

"citation_occurrence_id": "V2-SOL-03:s1",

"correction_ids": [],

"edition_id": "edition:keplinger-vor-2025",

"license": "CC BY-NC 4.0",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"raw_source_id": "s1",

"raw_title": "Assessment of Deep Research for dermatology literature reviews: Deep concern over the hype",

"readback": {

"captured_bytes": 5498,

"captured_sha256": "d74c1891ae09a2b89121f2a60ac837c89761a8086759e45db214ca68c1905cea",

"edition_id": "edition:keplinger-vor-2025",

"http_status": 403,

"media_type": "text/html; charset=UTF-8",

"redirect_chain": [],

"requested_url": "https://onlinelibrary.wiley.com/doi/10.1111/jdv.70035",

"resolved_url": "https://onlinelibrary.wiley.com/doi/10.1111/jdv.70035",

"retrieval_status": "inaccessible"

},

"requested_url": "https://onlinelibrary.wiley.com/doi/10.1111/jdv.70035",

"resolution_status": "unresolved",

"run_id": "V2-SOL-03",

"source_work_id": "work:keplinger-dermatology-audit",

"span_occurrence_ids": [

"V2-SOL-03:s1:s1_span1"

]

},

{

"citation_occurrence_id": "V2-SOL-03:s2",

"correction_ids": [],

"edition_id": "edition:drbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "s2",

"raw_title": "DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents",

"readback": {

"captured_bytes": 352002,

"captured_sha256": "9aa2894dbeaac30b23e7ffc8107a7f53b6e3855c8511838551de5bd7a422cc42",

"edition_id": "edition:drbench-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2506.11763",

"resolved_url": "https://arxiv.org/html/2506.11763",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2506.11763",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-03",

"source_work_id": "work:deepresearch-bench-paper",

"span_occurrence_ids": [

"V2-SOL-03:s2:s2_span1"

]

},

{

"citation_occurrence_id": "V2-SOL-03:s3",

"correction_ids": [],

"edition_id": "edition:deeptrace-arxiv-v1",

"license": "arXiv non-exclusive distribution license",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"raw_source_id": "s3",

"raw_title": "DeepTRACE: Auditing Deep Research AI Systems for Tracking Reliability Across Citations and Evidence",

"readback": {

"captured_bytes": 364455,

"captured_sha256": "0c28edecaedd882584e985caac1e14f4a03dd59e939e17ae755a3c5e07ae42b0",

"edition_id": "edition:deeptrace-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2509.04499",

"resolved_url": "https://arxiv.org/html/2509.04499",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2509.04499",

"resolution_status": "unresolved",

"run_id": "V2-SOL-03",

"source_work_id": "work:deeptrace",

"span_occurrence_ids": [

"V2-SOL-03:s3:s3_span1"

]

},

{

"citation_occurrence_id": "V2-SOL-03:s4",

"correction_ids": [],

"edition_id": "edition:url-health-arxiv-v1",

"license": "CC0 1.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "s4",

"raw_title": "Detecting and Correcting Reference Hallucinations in Commercial LLMs and Deep Research Agents",

"readback": {

"captured_bytes": 302498,

"captured_sha256": "332e5b5cb4b0ee7065b1bbc30436dfdfeafca8130fb87e32b0020026d8bbafe1",

"edition_id": "edition:url-health-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2604.03173",

"resolved_url": "https://arxiv.org/html/2604.03173",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2604.03173",

"resolution_status": "unresolved",

"run_id": "V2-SOL-03",

"source_work_id": "work:url-health",

"span_occurrence_ids": [

"V2-SOL-03:s4:s4_span1",

"V2-SOL-03:s4:s4_span2",

"V2-SOL-03:s4:s4_span3"

]

},

{

"citation_occurrence_id": "V2-SOL-03:s5",

"correction_ids": [],

"edition_id": "edition:cnv-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "s5",

"raw_title": "Cited but Not Verified: Parsing and Evaluating Source Attribution in LLM Deep Research Agents",

"readback": {

"captured_bytes": 138385,

"captured_sha256": "7c5e3c33f3122b07d176e6679cd6762a95a6babaf4a353686f019f905b5526ca",

"edition_id": "edition:cnv-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2605.06635",

"resolved_url": "https://arxiv.org/html/2605.06635",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2605.06635",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-03",

"source_work_id": "work:cited-not-verified",

"span_occurrence_ids": [

"V2-SOL-03:s5:s5_span1",

"V2-SOL-03:s5:s5_span2"

]

},

{

"citation_occurrence_id": "V2-SOL-03:s6",

"correction_ids": [],

"edition_id": "edition:citation-verifier-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "s6",

"raw_title": "Do You Need a Frontier Model as a Citation Verifier? Benchmarking Rubric LLMs for Deep-Research Source Attribution",

"readback": {

"captured_bytes": 43218,

"captured_sha256": "c3efca0e3c20c437e63b3a990c85b2392f0244d3dc8194312a8394923a16cbec",

"edition_id": "edition:citation-verifier-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/abs/2607.08700",

"resolved_url": "https://arxiv.org/abs/2607.08700",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/abs/2607.08700",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-03",

"source_work_id": "work:citation-verifier-benchmark",

"span_occurrence_ids": [

"V2-SOL-03:s6:s6_span1"

]

},

{

"citation_occurrence_id": "V2-SOL-03:s7",

"correction_ids": [],

"edition_id": "edition:drbench-repo-main-469cce5",

"license": "Apache-2.0",

"license_treatment": "metadata and quote-minimal README spans",

"raw_source_id": "s7",

"raw_title": "DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents",

"readback": {

"captured_bytes": 373533,

"captured_sha256": "ec9a4efdb736c6b5ec198cfbc5534eb67b5d421efa12cf795d7990be6c1e3b54",

"edition_id": "edition:drbench-repo-main-469cce5",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://github.com/Ayanami0730/deep_research_bench",

"resolved_url": "https://github.com/Ayanami0730/deep_research_bench",

"retrieval_status": "retrieved"

},

"requested_url": "https://github.com/Ayanami0730/deep_research_bench",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-03",

"source_work_id": "work:deepresearch-bench-repository",

"span_occurrence_ids": [

"V2-SOL-03:s7:s7_span1"

]

},

{

"citation_occurrence_id": "V2-SOL-04:S1",

"correction_ids": [],

"edition_id": "edition:cnv-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S1",

"raw_title": "Cited but Not Verified: Parsing and Evaluating Source Attribution in LLM Deep Research Agents",

"readback": {

"captured_bytes": 138385,

"captured_sha256": "7c5e3c33f3122b07d176e6679cd6762a95a6babaf4a353686f019f905b5526ca",

"edition_id": "edition:cnv-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2605.06635v1",

"resolved_url": "https://arxiv.org/html/2605.06635v1",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2605.06635v1",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-04",

"source_work_id": "work:cited-not-verified",

"span_occurrence_ids": [

"V2-SOL-04:S1:S1-SP1",

"V2-SOL-04:S1:S1-SP2",

"V2-SOL-04:S1:S1-SP3"

]

},

{

"citation_occurrence_id": "V2-SOL-04:S2",

"correction_ids": [],

"edition_id": "edition:drbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S2",

"raw_title": "DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents",

"readback": {

"captured_bytes": 352002,

"captured_sha256": "9aa2894dbeaac30b23e7ffc8107a7f53b6e3855c8511838551de5bd7a422cc42",

"edition_id": "edition:drbench-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2506.11763v1",

"resolved_url": "https://arxiv.org/html/2506.11763v1",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2506.11763v1",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-04",

"source_work_id": "work:deepresearch-bench-paper",

"span_occurrence_ids": [

"V2-SOL-04:S2:S2-SP1",

"V2-SOL-04:S2:S2-SP2",

"V2-SOL-04:S2:S2-SP3",

"V2-SOL-04:S2:S2-SP4"

]

},

{

"citation_occurrence_id": "V2-SOL-04:S3",

"correction_ids": [],

"edition_id": "edition:reportbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S3",

"raw_title": "ReportBench: Evaluating Deep Research Agents via Academic Survey Tasks",

"readback": {

"captured_bytes": 157386,

"captured_sha256": "055e568189a402dd510c7c84be60f59465d9a001808936bc25cb0026ac25d267",

"edition_id": "edition:reportbench-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2508.15804v1",

"resolved_url": "https://arxiv.org/html/2508.15804v1",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2508.15804v1",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-04",

"source_work_id": "work:reportbench",

"span_occurrence_ids": [

"V2-SOL-04:S3:S3-SP1",

"V2-SOL-04:S3:S3-SP2"

]

},

{

"citation_occurrence_id": "V2-SOL-04:S4",

"correction_ids": [],

"edition_id": "edition:deeptrace-arxiv-v1",

"license": "arXiv non-exclusive distribution license",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"raw_source_id": "S4",

"raw_title": "DeepTRACE: Auditing Deep Research AI Systems for Tracking Reliability Across Citations and Evidence",

"readback": {

"captured_bytes": 364455,

"captured_sha256": "0c28edecaedd882584e985caac1e14f4a03dd59e939e17ae755a3c5e07ae42b0",

"edition_id": "edition:deeptrace-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2509.04499v1",

"resolved_url": "https://arxiv.org/html/2509.04499v1",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2509.04499v1",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-04",

"source_work_id": "work:deeptrace",

"span_occurrence_ids": [

"V2-SOL-04:S4:S4-SP1",

"V2-SOL-04:S4:S4-SP2"

]

},

{

"citation_occurrence_id": "V2-SOL-04:S5",

"correction_ids": [],

"edition_id": "edition:keplinger-vor-2025",

"license": "CC BY-NC 4.0",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"raw_source_id": "S5",

"raw_title": "Assessment of Deep Research for dermatology literature reviews: Deep concern over the hype",

"readback": {

"captured_bytes": 111993,

"captured_sha256": "abbaa9f22035230d13152a68aaeecd8e83e40867c9dae4acd1d547e36c866793",

"edition_id": "edition:keplinger-vor-2025",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"resolved_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"retrieval_status": "retrieved"

},

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-04",

"source_work_id": "work:keplinger-dermatology-audit",

"span_occurrence_ids": [

"V2-SOL-04:S5:S5-SP1",

"V2-SOL-04:S5:S5-SP2",

"V2-SOL-04:S5:S5-SP3"

]

},

{

"citation_occurrence_id": "V2-SOL-04:S6",

"correction_ids": [

"correction:mendeley-file-readback"

],

"edition_id": "edition:keplinger-supplement-v1",

"license": "CC BY 4.0",

"license_treatment": "metadata and quote-minimal landing-page spans",

"raw_source_id": "S6",

"raw_title": "Supplementary materials of the article: Assessment of Deep Research for Dermatology Literature Reviews: Deep Concern Over the Hype",

"readback": {

"captured_bytes": 115612,

"captured_sha256": "da6308937337b2dfd3f97dc62f817e4a7ec362329ede0fce4c9bcdfd99ad93b1",

"edition_id": "edition:keplinger-supplement-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://data.mendeley.com/datasets/3s73z9zf3c/1",

"resolved_url": "https://data.mendeley.com/datasets/3s73z9zf3c/1",

"retrieval_status": "retrieved"

},

"requested_url": "https://data.mendeley.com/datasets/3s73z9zf3c/1",

"resolution_status": "unresolved",

"run_id": "V2-SOL-04",

"source_work_id": "work:keplinger-supplement",

"span_occurrence_ids": [

"V2-SOL-04:S6:S6-SP1"

]

},

{

"citation_occurrence_id": "V2-SOL-04:S7",

"correction_ids": [],

"edition_id": "edition:citation-verifier-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S7",

"raw_title": "Do You Need a Frontier Model as a Citation Verifier? Benchmarking Rubric LLMs for Deep-Research Source Attribution",

"readback": {

"captured_bytes": 43218,

"captured_sha256": "c3efca0e3c20c437e63b3a990c85b2392f0244d3dc8194312a8394923a16cbec",

"edition_id": "edition:citation-verifier-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/abs/2607.08700",

"resolved_url": "https://arxiv.org/abs/2607.08700",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/abs/2607.08700",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-SOL-04",

"source_work_id": "work:citation-verifier-benchmark",

"span_occurrence_ids": [

"V2-SOL-04:S7:S7-SP1",

"V2-SOL-04:S7:S7-SP2"

]

},

{

"citation_occurrence_id": "V2-TERRA-01:s1_deepresearchbench",

"correction_ids": [],

"edition_id": "edition:drbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "s1_deepresearchbench",

"raw_title": "DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents",

"readback": {

"captured_bytes": 3678399,

"captured_sha256": "8fbf30398f5e62f8839f0c9c8609bbb9e3cd0b57ae27d4bf33cb5db2007d1118",

"edition_id": "edition:drbench-arxiv-v1",

"http_status": 200,

"media_type": "application/pdf",

"redirect_chain": [],

"requested_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"resolved_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"retrieval_status": "retrieved"

},

"requested_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-01",

"source_work_id": "work:deepresearch-bench-paper",

"span_occurrence_ids": [

"V2-TERRA-01:s1_deepresearchbench:s1_method",

"V2-TERRA-01:s1_deepresearchbench:s1_table",

"V2-TERRA-01:s1_deepresearchbench:s1_dates"

]

},

{

"citation_occurrence_id": "V2-TERRA-01:s2_reportbench",

"correction_ids": [],

"edition_id": "edition:reportbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "s2_reportbench",

"raw_title": "ReportBench: Evaluating Deep Research Agents via Academic Survey Tasks",

"readback": {

"captured_bytes": 729607,

"captured_sha256": "90730ad75011d460305caf45a5be33b1b5eb4126f7e7efc1506f63beda1f91d3",

"edition_id": "edition:reportbench-arxiv-v1",

"http_status": 200,

"media_type": "application/pdf",

"redirect_chain": [],

"requested_url": "https://arxiv.org/pdf/2508.15804",

"resolved_url": "https://arxiv.org/pdf/2508.15804",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/pdf/2508.15804",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-01",

"source_work_id": "work:reportbench",

"span_occurrence_ids": [

"V2-TERRA-01:s2_reportbench:s2_method",

"V2-TERRA-01:s2_reportbench:s2_metric_table",

"V2-TERRA-01:s2_reportbench:s2_collection"

]

},

{

"citation_occurrence_id": "V2-TERRA-01:s3_researcherbench",

"correction_ids": [],

"edition_id": "edition:researcherbench-arxiv-v1",

"license": "arXiv non-exclusive distribution license",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"raw_source_id": "s3_researcherbench",

"raw_title": "ResearcherBench: Evaluating Deep AI Research Systems on the Frontiers of Scientific Inquiry",

"readback": {

"captured_bytes": 1566914,

"captured_sha256": "1571435270d3e781ab8d5254913af00a7dcf0efcb99ec8e6d3db87a7b5c92f02",

"edition_id": "edition:researcherbench-arxiv-v1",

"http_status": 200,

"media_type": "application/pdf",

"redirect_chain": [],

"requested_url": "https://arxiv.org/pdf/2507.16280",

"resolved_url": "https://arxiv.org/pdf/2507.16280",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/pdf/2507.16280",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-01",

"source_work_id": "work:researcherbench",

"span_occurrence_ids": [

"V2-TERRA-01:s3_researcherbench:s3_definition",

"V2-TERRA-01:s3_researcherbench:s3_models_time",

"V2-TERRA-01:s3_researcherbench:s3_table"

]

},

{

"citation_occurrence_id": "V2-TERRA-01:s4_cited_not_verified",

"correction_ids": [],

"edition_id": "edition:cnv-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "s4_cited_not_verified",

"raw_title": "Cited but Not Verified: Parsing and Evaluating Source Attribution in LLM Deep Research Agents",

"readback": {

"captured_bytes": 1424589,

"captured_sha256": "db5b7b7e3d9ce9fc6d6713b60f65f0d81b07090ded2a28be7f541d1c290230b5",

"edition_id": "edition:cnv-arxiv-v1",

"http_status": 200,

"media_type": "application/pdf",

"redirect_chain": [],

"requested_url": "https://arxiv.org/pdf/2605.06635",

"resolved_url": "https://arxiv.org/pdf/2605.06635",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/pdf/2605.06635",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-01",

"source_work_id": "work:cited-not-verified",

"span_occurrence_ids": [

"V2-TERRA-01:s4_cited_not_verified:s4_definition_table",

"V2-TERRA-01:s4_cited_not_verified:s4_depth"

]

},

{

"citation_occurrence_id": "V2-TERRA-01:s5_urlhealth",

"correction_ids": [],

"edition_id": "edition:url-health-arxiv-v1",

"license": "CC0 1.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "s5_urlhealth",

"raw_title": "Detecting and Correcting Reference Hallucinations in Commercial LLMs and Deep Research Agents",

"readback": {

"captured_bytes": 416944,

"captured_sha256": "e8145a9f62f2a2bf0c5f3e2607160c56e010d02da965b5f65dbfa8f07ddde06c",

"edition_id": "edition:url-health-arxiv-v1",

"http_status": 200,

"media_type": "application/pdf",

"redirect_chain": [],

"requested_url": "https://arxiv.org/pdf/2604.03173",

"resolved_url": "https://arxiv.org/pdf/2604.03173",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/pdf/2604.03173",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-01",

"source_work_id": "work:url-health",

"span_occurrence_ids": [

"V2-TERRA-01:s5_urlhealth:s5_scope",

"V2-TERRA-01:s5_urlhealth:s5_table",

"V2-TERRA-01:s5_urlhealth:s5_comparison"

]

},

{

"citation_occurrence_id": "V2-TERRA-02:s1_deepresearchbench",

"correction_ids": [],

"edition_id": "edition:drbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "s1_deepresearchbench",

"raw_title": "DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents",

"readback": {

"captured_bytes": 3678399,

"captured_sha256": "8fbf30398f5e62f8839f0c9c8609bbb9e3cd0b57ae27d4bf33cb5db2007d1118",

"edition_id": "edition:drbench-arxiv-v1",

"http_status": 200,

"media_type": "application/pdf",

"redirect_chain": [],

"requested_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"resolved_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"retrieval_status": "retrieved"

},

"requested_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-02",

"source_work_id": "work:deepresearch-bench-paper",

"span_occurrence_ids": [

"V2-TERRA-02:s1_deepresearchbench:s1_fact_method",

"V2-TERRA-02:s1_deepresearchbench:s1_fact_results",

"V2-TERRA-02:s1_deepresearchbench:s1_fact_formula",

"V2-TERRA-02:s1_deepresearchbench:s1_fact_judge_validation"

]

},

{

"citation_occurrence_id": "V2-TERRA-02:s2_liveresearchbench",

"correction_ids": [

"correction:liveresearchbench-license"

],

"edition_id": "edition:liveresearchbench-arxiv-v5",

"license": "CC BY-NC-SA 4.0",

"license_treatment": "quote-minimal attributed spans; raw CC BY claim is corrected in review records",

"raw_source_id": "s2_liveresearchbench",

"raw_title": "LiveResearchBench: A Live Benchmark for User-Centric Deep Research in the Wild",

"readback": {

"captured_bytes": 13057891,

"captured_sha256": "579b9728b76cfd242e9c94d9ff2985e196bbc72b5a741030e4f308ede04a4f69",

"edition_id": "edition:liveresearchbench-arxiv-v5",

"http_status": 200,

"media_type": "application/pdf",

"redirect_chain": [],

"requested_url": "https://arxiv.org/pdf/2510.14240v5",

"resolved_url": "https://arxiv.org/pdf/2510.14240v5",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/pdf/2510.14240v5",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-02",

"source_work_id": "work:liveresearchbench",

"span_occurrence_ids": [

"V2-TERRA-02:s2_liveresearchbench:s2_rubric_tree",

"V2-TERRA-02:s2_liveresearchbench:s2_table7",

"V2-TERRA-02:s2_liveresearchbench:s2_validation"

]

},

{

"citation_occurrence_id": "V2-TERRA-02:s3_deeptrace",

"correction_ids": [],

"edition_id": "edition:deeptrace-iclr-2026",

"license": "unknown",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"raw_source_id": "s3_deeptrace",

"raw_title": "DeepTRACE: Auditing Deep Research AI Systems for Tracking Reliability Across Citations and Evidence",

"readback": {

"captured_bytes": 2982839,

"captured_sha256": "dea4981c1066d0240a005f603b3b14419c7e32beb5191b3df847ceb26af3d6b6",

"edition_id": "edition:deeptrace-iclr-2026",

"http_status": 200,

"media_type": "application/pdf",

"redirect_chain": [],

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"resolved_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"retrieval_status": "retrieved"

},

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-02",

"source_work_id": "work:deeptrace",

"span_occurrence_ids": [

"V2-TERRA-02:s3_deeptrace:s3_definition",

"V2-TERRA-02:s3_deeptrace:s3_table1",

"V2-TERRA-02:s3_deeptrace:s3_corpus_and_retrieval",

"V2-TERRA-02:s3_deeptrace:s3_judge_validation"

]

},

{

"citation_occurrence_id": "V2-TERRA-02:s4_researcherbench",

"correction_ids": [],

"edition_id": "edition:researcherbench-arxiv-v1",

"license": "arXiv non-exclusive distribution license",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"raw_source_id": "s4_researcherbench",

"raw_title": "ResearcherBench: Evaluating Deep AI Research Systems on the Frontiers of Scientific Inquiry",

"readback": {

"captured_bytes": 280402,

"captured_sha256": "c080d7304274d70a39651f49297cc65910ca833d7ee6a3de3052370733e44ee3",

"edition_id": "edition:researcherbench-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2507.16280",

"resolved_url": "https://arxiv.org/html/2507.16280",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2507.16280",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-02",

"source_work_id": "work:researcherbench",

"span_occurrence_ids": [

"V2-TERRA-02:s4_researcherbench:s4_method",

"V2-TERRA-02:s4_researcherbench:s4_results"

]

},

{

"citation_occurrence_id": "V2-TERRA-03:S1",

"correction_ids": [],

"edition_id": "edition:cnv-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S1",

"raw_title": "Cited but Not Verified: Parsing and Evaluating Source Attribution in LLM Deep Research Agents",

"readback": {

"captured_bytes": 138385,

"captured_sha256": "7c5e3c33f3122b07d176e6679cd6762a95a6babaf4a353686f019f905b5526ca",

"edition_id": "edition:cnv-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2605.06635",

"resolved_url": "https://arxiv.org/html/2605.06635",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2605.06635",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-03",

"source_work_id": "work:cited-not-verified",

"span_occurrence_ids": [

"V2-TERRA-03:S1:S1-A",

"V2-TERRA-03:S1:S1-B",

"V2-TERRA-03:S1:S1-C",

"V2-TERRA-03:S1:S1-D",

"V2-TERRA-03:S1:S1-E",

"V2-TERRA-03:S1:S1-F"

]

},

{

"citation_occurrence_id": "V2-TERRA-03:S2",

"correction_ids": [],

"edition_id": "edition:drbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S2",

"raw_title": "DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents",

"readback": {

"captured_bytes": 352002,

"captured_sha256": "9aa2894dbeaac30b23e7ffc8107a7f53b6e3855c8511838551de5bd7a422cc42",

"edition_id": "edition:drbench-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2506.11763",

"resolved_url": "https://arxiv.org/html/2506.11763",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2506.11763",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-03",

"source_work_id": "work:deepresearch-bench-paper",

"span_occurrence_ids": [

"V2-TERRA-03:S2:S2-A",

"V2-TERRA-03:S2:S2-B",

"V2-TERRA-03:S2:S2-C",

"V2-TERRA-03:S2:S2-D"

]

},

{

"citation_occurrence_id": "V2-TERRA-03:S3",

"correction_ids": [],

"edition_id": "edition:reportbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S3",

"raw_title": "ReportBench: Evaluating Deep Research Agents via Academic Survey Tasks",

"readback": {

"captured_bytes": 157386,

"captured_sha256": "055e568189a402dd510c7c84be60f59465d9a001808936bc25cb0026ac25d267",

"edition_id": "edition:reportbench-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2508.15804",

"resolved_url": "https://arxiv.org/html/2508.15804",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2508.15804",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-03",

"source_work_id": "work:reportbench",

"span_occurrence_ids": [

"V2-TERRA-03:S3:S3-A",

"V2-TERRA-03:S3:S3-B",

"V2-TERRA-03:S3:S3-C",

"V2-TERRA-03:S3:S3-D"

]

},

{

"citation_occurrence_id": "V2-TERRA-03:S4",

"correction_ids": [],

"edition_id": "edition:keplinger-vor-2025",

"license": "CC BY-NC 4.0",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"raw_source_id": "S4",

"raw_title": "Assessment of Deep Research for dermatology literature reviews: Deep concern over the hype",

"readback": {

"captured_bytes": 111993,

"captured_sha256": "abbaa9f22035230d13152a68aaeecd8e83e40867c9dae4acd1d547e36c866793",

"edition_id": "edition:keplinger-vor-2025",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"resolved_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"retrieval_status": "retrieved"

},

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-03",

"source_work_id": "work:keplinger-dermatology-audit",

"span_occurrence_ids": [

"V2-TERRA-03:S4:S4-A",

"V2-TERRA-03:S4:S4-B",

"V2-TERRA-03:S4:S4-C"

]

},

{

"citation_occurrence_id": "V2-TERRA-03:S5",

"correction_ids": [],

"edition_id": "edition:drbench-repo-main-469cce5",

"license": "Apache-2.0",

"license_treatment": "metadata and quote-minimal README spans",

"raw_source_id": "S5",

"raw_title": "DeepResearch Bench official repository",

"readback": {

"captured_bytes": 373533,

"captured_sha256": "ec9a4efdb736c6b5ec198cfbc5534eb67b5d421efa12cf795d7990be6c1e3b54",

"edition_id": "edition:drbench-repo-main-469cce5",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://github.com/Ayanami0730/deep_research_bench",

"resolved_url": "https://github.com/Ayanami0730/deep_research_bench",

"retrieval_status": "retrieved"

},

"requested_url": "https://github.com/Ayanami0730/deep_research_bench",

"resolution_status": "resolved-and-span-matched",

"run_id": "V2-TERRA-03",

"source_work_id": "work:deepresearch-bench-repository",

"span_occurrence_ids": [

"V2-TERRA-03:S5:S5-A",

"V2-TERRA-03:S5:S5-B"

]

},

{

"citation_occurrence_id": "V2-TERRA-03:S6",

"correction_ids": [],

"edition_id": "edition:keplinger-vor-2025",

"license": "CC BY-NC 4.0",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"raw_source_id": "S6",

"raw_title": "PubMed record and Figure 1 caption for Assessment of Deep Research for dermatology literature reviews: Deep concern over the hype",

"readback": {

"captured_bytes": 5565,

"captured_sha256": "a46109544fe4ff4504fab5e97abea3cb7172367aba54a308937386394f0ff046",

"edition_id": "edition:keplinger-vor-2025",

"http_status": 203,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://pubmed.ncbi.nlm.nih.gov/40904191/",

"resolved_url": "https://pubmed.ncbi.nlm.nih.gov/40904191/",

"retrieval_status": "inaccessible"

},

"requested_url": "https://pubmed.ncbi.nlm.nih.gov/40904191/",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-03",

"source_work_id": "work:keplinger-dermatology-audit",

"span_occurrence_ids": [

"V2-TERRA-03:S6:S6-A"

]

},

{

"citation_occurrence_id": "V2-TERRA-04:S1_deepresearchbench",

"correction_ids": [],

"edition_id": "edition:drbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S1_deepresearchbench",

"raw_title": "DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents",

"readback": {

"captured_bytes": 352002,

"captured_sha256": "9aa2894dbeaac30b23e7ffc8107a7f53b6e3855c8511838551de5bd7a422cc42",

"edition_id": "edition:drbench-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2506.11763",

"resolved_url": "https://arxiv.org/html/2506.11763",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2506.11763",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-04",

"source_work_id": "work:deepresearch-bench-paper",

"span_occurrence_ids": [

"V2-TERRA-04:S1_deepresearchbench:S1_method",

"V2-TERRA-04:S1_deepresearchbench:S1_table1",

"V2-TERRA-04:S1_deepresearchbench:S1_metric",

"V2-TERRA-04:S1_deepresearchbench:S1_time"

]

},

{

"citation_occurrence_id": "V2-TERRA-04:S2_researcherbench",

"correction_ids": [],

"edition_id": "edition:researcherbench-arxiv-v1",

"license": "arXiv non-exclusive distribution license",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"raw_source_id": "S2_researcherbench",

"raw_title": "ResearcherBench: Evaluating Deep AI Research Systems on the Frontiers of Scientific Inquiry",

"readback": {

"captured_bytes": 280402,

"captured_sha256": "c080d7304274d70a39651f49297cc65910ca833d7ee6a3de3052370733e44ee3",

"edition_id": "edition:researcherbench-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2507.16280",

"resolved_url": "https://arxiv.org/html/2507.16280",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2507.16280",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-04",

"source_work_id": "work:researcherbench",

"span_occurrence_ids": [

"V2-TERRA-04:S2_researcherbench:S2_method",

"V2-TERRA-04:S2_researcherbench:S2_table2",

"V2-TERRA-04:S2_researcherbench:S2_scope",

"V2-TERRA-04:S2_researcherbench:S2_time"

]

},

{

"citation_occurrence_id": "V2-TERRA-04:S3_reportbench",

"correction_ids": [],

"edition_id": "edition:reportbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S3_reportbench",

"raw_title": "ReportBench: Evaluating Deep Research Agents via Academic Survey Tasks",

"readback": {

"captured_bytes": 157386,

"captured_sha256": "055e568189a402dd510c7c84be60f59465d9a001808936bc25cb0026ac25d267",

"edition_id": "edition:reportbench-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2508.15804",

"resolved_url": "https://arxiv.org/html/2508.15804",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2508.15804",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-04",

"source_work_id": "work:reportbench",

"span_occurrence_ids": [

"V2-TERRA-04:S3_reportbench:S3_method",

"V2-TERRA-04:S3_reportbench:S3_table1",

"V2-TERRA-04:S3_reportbench:S3_time",

"V2-TERRA-04:S3_reportbench:S3_limitations"

]

},

{

"citation_occurrence_id": "V2-TERRA-04:S4_urlhealth",

"correction_ids": [],

"edition_id": "edition:url-health-arxiv-v1",

"license": "CC0 1.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S4_urlhealth",

"raw_title": "Detecting and Correcting Reference Hallucinations in Commercial LLMs and Deep Research Agents",

"readback": {

"captured_bytes": 302498,

"captured_sha256": "332e5b5cb4b0ee7065b1bbc30436dfdfeafca8130fb87e32b0020026d8bbafe1",

"edition_id": "edition:url-health-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2604.03173",

"resolved_url": "https://arxiv.org/html/2604.03173",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2604.03173",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-04",

"source_work_id": "work:url-health",

"span_occurrence_ids": [

"V2-TERRA-04:S4_urlhealth:S4_table2",

"V2-TERRA-04:S4_urlhealth:S4_comparison",

"V2-TERRA-04:S4_urlhealth:S4_method",

"V2-TERRA-04:S4_urlhealth:S4_limitations"

]

},

{

"citation_occurrence_id": "V2-TERRA-04:S5_cited_not_verified",

"correction_ids": [],

"edition_id": "edition:cnv-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"raw_source_id": "S5_cited_not_verified",

"raw_title": "Cited but Not Verified: Parsing and Evaluating Source Attribution in LLM Deep Research Agents",

"readback": {

"captured_bytes": 138385,

"captured_sha256": "7c5e3c33f3122b07d176e6679cd6762a95a6babaf4a353686f019f905b5526ca",

"edition_id": "edition:cnv-arxiv-v1",

"http_status": 200,

"media_type": "text/html; charset=utf-8",

"redirect_chain": [],

"requested_url": "https://arxiv.org/html/2605.06635",

"resolved_url": "https://arxiv.org/html/2605.06635",

"retrieval_status": "retrieved"

},

"requested_url": "https://arxiv.org/html/2605.06635",

"resolution_status": "unresolved",

"run_id": "V2-TERRA-04",

"source_work_id": "work:cited-not-verified",

"span_occurrence_ids": [

"V2-TERRA-04:S5_cited_not_verified:S5_method",

"V2-TERRA-04:S5_cited_not_verified:S5_scope",

"V2-TERRA-04:S5_cited_not_verified:S5_table1",

"V2-TERRA-04:S5_cited_not_verified:S5_depth",

"V2-TERRA-04:S5_cited_not_verified:S5_limitations"

]

}

],

"claims": [

{

"canonical_proposition": "In one five-system dermatology-review audit, reference identifiability and metadata correctness did not establish sentence-level claim-citation concordance.",

"claim_occurrence_id": "V2-SOL-01:r1_human_dermatology_chatgpt",

"correction_ids": [],

"raw_proposition": "In three human-reviewed dermatology literature-review runs, ChatGPT Deep Research usually produced identifiable references, but roughly half of citation-bearing sentences contained at least one claim-citation inaccuracy.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-01:s1_keplinger",

"V2-SOL-01:s1a_keplinger_data"

],

"span_occurrence_ids": [

"V2-SOL-01:s1_keplinger:sp1a",

"V2-SOL-01:s1_keplinger:sp1b",

"V2-SOL-01:s1_keplinger:sp1c"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:keplinger-metadata-versus-support"

},

{

"canonical_proposition": "In one five-system dermatology-review audit, reference identifiability and metadata correctness did not establish sentence-level claim-citation concordance.",

"claim_occurrence_id": "V2-SOL-01:r2_human_dermatology_lechat",

"correction_ids": [],

"raw_proposition": "The same human-reviewed experiment found a similar gap for Le Chat Think: mostly identifiable references but frequent inaccuracies in citation-bearing sentences.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-01:s1_keplinger",

"V2-SOL-01:s1a_keplinger_data"

],

"span_occurrence_ids": [

"V2-SOL-01:s1_keplinger:sp1a",

"V2-SOL-01:s1_keplinger:sp1b",

"V2-SOL-01:s1_keplinger:sp1c"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:keplinger-metadata-versus-support"

},

{

"canonical_proposition": "DeepResearch Bench evaluated binary support for deduplicated statement-URL pairs; numeric values are edition-specific.",

"claim_occurrence_id": "V2-SOL-01:r3_deepresearch_bench_fact",

"correction_ids": [

"correction:drbench-edition-drift"

],

"raw_proposition": "DeepResearch Bench measured whether retrieved webpage text supported deduplicated statement-URL pairs and reported materially different task-macro citation accuracy across named agents.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-01:s2_drbench",

"V2-SOL-01:s2a_drbench_repo"

],

"span_occurrence_ids": [

"V2-SOL-01:s2_drbench:sp2a",

"V2-SOL-01:s2_drbench:sp2b",

"V2-SOL-01:s2_drbench:sp2c",

"V2-SOL-01:s2_drbench:sp2d"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deepresearch-bench-fact"

},

{

"canonical_proposition": "DeepTRACE measured statement-source support and found between-system variation; its Gemini value is internally inconsistent between table and prose.",

"claim_occurrence_id": "V2-SOL-01:r4_deeptrace_support",

"correction_ids": [

"correction:deeptrace-gemini-number"

],

"raw_proposition": "DeepTRACE separately measured whether citations pointed to sources that supported the cited statements and whether relevant statements were supported by any listed source.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-01:s3_deeptrace"

],

"span_occurrence_ids": [

"V2-SOL-01:s3_deeptrace:sp3a",

"V2-SOL-01:s3_deeptrace:sp3b",

"V2-SOL-01:s3_deeptrace:sp3c",

"V2-SOL-01:s3_deeptrace:sp3d"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deeptrace-support-variation"

},

{

"canonical_proposition": "Cited but Not Verified operationalized link access, topical relevance, and factual support separately and reported materially different rates.",

"claim_occurrence_id": "V2-SOL-01:r5_cited_but_not_verified_cross_model",

"correction_ids": [],

"raw_proposition": "Cited but Not Verified found that working and topically relevant links did not imply that the linked content supported the attributed facts.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-01:s4_cited_not_verified"

],

"span_occurrence_ids": [

"V2-SOL-01:s4_cited_not_verified:sp4a",

"V2-SOL-01:s4_cited_not_verified:sp4b"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-link-relevance-support-gap"

},

{

"canonical_proposition": "Within the Cited but Not Verified harness, the 2-to-150-call ablation reduced reported Fact Check scores for two setups without establishing a general causal law about search depth.",

"claim_occurrence_id": "V2-SOL-01:r6_cited_but_not_verified_depth_ablation",

"correction_ids": [],

"raw_proposition": "Within the same evaluation pipeline, increasing the permitted tool-call count from 2 to 150 was associated with lower Fact Check accuracy for two frontier-model agent setups while link and topical-relevance rates remained high.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-01:s4_cited_not_verified"

],

"span_occurrence_ids": [

"V2-SOL-01:s4_cited_not_verified:sp4b",

"V2-SOL-01:s4_cited_not_verified:sp4c"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-search-depth-ablation"

},

{

"canonical_proposition": "The URL-health study measured HTTP resolution and Wayback presence, not semantic claim support, on outputs including reused DeepResearch Bench material.",

"claim_occurrence_id": "V2-SOL-01:r7_url_resolution_drbench",

"correction_ids": [],

"raw_proposition": "A separate URL-health study measured whether citation URLs resolved and whether non-resolving URLs had any Wayback Machine record; it did not test whether resolved pages supported the associated claims.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-01:s5_url_health"

],

"span_occurrence_ids": [

"V2-SOL-01:s5_url_health:sp5a",

"V2-SOL-01:s5_url_health:sp5b",

"V2-SOL-01:s5_url_health:sp5c"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:url-health-resolution"

},

{

"canonical_proposition": "The URL-health study measured HTTP resolution and Wayback presence, not semantic claim support, on outputs including reused DeepResearch Bench material.",

"claim_occurrence_id": "V2-SOL-02:R1_url_resolution",

"correction_ids": [],

"raw_proposition": "Deep-research-agent citation URLs sometimes fail to resolve, and some non-resolving URLs have no Wayback record.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-02:S1"

],

"span_occurrence_ids": [

"V2-SOL-02:S1:S1_SPAN_1",

"V2-SOL-02:S1:S1_SPAN_2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:url-health-resolution"

},

{

"canonical_proposition": "In one five-system dermatology-review audit, reference identifiability and metadata correctness did not establish sentence-level claim-citation concordance.",

"claim_occurrence_id": "V2-SOL-02:R2_human_dermatology_claim_support",

"correction_ids": [],

"raw_proposition": "Human manual inspection found frequent subtle inaccuracies in claims attached to citations even when reference lists were largely identifiable.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-02:S2",

"V2-SOL-02:S3"

],

"span_occurrence_ids": [

"V2-SOL-02:S2:S2_SPAN_1",

"V2-SOL-02:S3:S3_SPAN_1",

"V2-SOL-02:S3:S3_SPAN_2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:keplinger-metadata-versus-support"

},

{

"canonical_proposition": "DeepResearch Bench evaluated binary support for deduplicated statement-URL pairs; numeric values are edition-specific.",

"claim_occurrence_id": "V2-SOL-02:R3_deepresearch_bench_fact",

"correction_ids": [

"correction:drbench-edition-drift"

],

"raw_proposition": "The final ICLR edition of DeepResearch Bench found that a majority, but not all, unique statement-URL pairs were judged supported.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-02:S4"

],

"span_occurrence_ids": [

"V2-SOL-02:S4:S4_SPAN_1",

"V2-SOL-02:S4:S4_SPAN_2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deepresearch-bench-fact"

},

{

"canonical_proposition": "ReportBench compared cited statements with retrieved cited-page content and reported sub-100-percent semantic match rates in its bounded survey-task setting.",

"claim_occurrence_id": "V2-SOL-02:R4_reportbench_match_rate",

"correction_ids": [],

"raw_proposition": "On academic-survey tasks, an automated source-retrieval and semantic-consistency pipeline found citation-source match rates below 80% for two commercial deep-research products.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-02:S5"

],

"span_occurrence_ids": [

"V2-SOL-02:S5:S5_SPAN_1",

"V2-SOL-02:S5:S5_SPAN_2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:reportbench-match-rate"

},

{

"canonical_proposition": "DeepTRACE measured statement-source support and found between-system variation; its Gemini value is internally inconsistent between table and prose.",

"claim_occurrence_id": "V2-SOL-02:R5_deeptrace_audit",

"correction_ids": [

"correction:independent-semantic-warrant-gaps"

],

"raw_proposition": "An ICLR 2026 audit found large between-system variation in both unsupported statements and citation accuracy among deep-research configurations.",

"review_status": "independent-review-no-credit-insufficient-span-semantics",

"source_citation_occurrence_ids": [

"V2-SOL-02:S6"

],

"span_occurrence_ids": [

"V2-SOL-02:S6:S6_SPAN_1",

"V2-SOL-02:S6:S6_SPAN_2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deeptrace-support-variation"

},

{

"canonical_proposition": "Cited but Not Verified operationalized link access, topical relevance, and factual support separately and reported materially different rates.",

"claim_occurrence_id": "V2-SOL-02:R6_surface_quality_vs_fact_support",

"correction_ids": [],

"raw_proposition": "A 2026 preprint found that working and topically relevant links did not imply that the attributed factual claim was supported.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-02:S7"

],

"span_occurrence_ids": [

"V2-SOL-02:S7:S7_SPAN_1"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-link-relevance-support-gap"

},

{

"canonical_proposition": "Within the Cited but Not Verified harness, the 2-to-150-call ablation reduced reported Fact Check scores for two setups without establishing a general causal law about search depth.",

"claim_occurrence_id": "V2-SOL-02:R7_search_depth_ablation",

"correction_ids": [],

"raw_proposition": "In a controlled ablation, greater allowed search depth did not improve citation support and was associated with lower Fact Check accuracy.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-02:S7"

],

"span_occurrence_ids": [

"V2-SOL-02:S7:S7_SPAN_2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-search-depth-ablation"

},

{

"canonical_proposition": "In one five-system dermatology-review audit, reference identifiability and metadata correctness did not establish sentence-level claim-citation concordance.",

"claim_occurrence_id": "V2-SOL-03:r1_reference_metadata_human_audit",

"correction_ids": [],

"raw_proposition": "In a three-run dermatology review task, ChatGPT Deep Research and Le Chat mostly produced identifiable references, while several other agents frequently fabricated authors or titles.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-03:s1"

],

"span_occurrence_ids": [

"V2-SOL-03:s1:s1_span1"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:keplinger-metadata-versus-support"

},

{

"canonical_proposition": "In one five-system dermatology-review audit, reference identifiability and metadata correctness did not establish sentence-level claim-citation concordance.",

"claim_occurrence_id": "V2-SOL-03:r2_claim_support_human_audit",

"correction_ids": [],

"raw_proposition": "For the two systems with the best reference-list performance, more than half of citation-bearing sentences contained at least one subtle claim-citation inaccuracy.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-03:s1"

],

"span_occurrence_ids": [

"V2-SOL-03:s1:s1_span1"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:keplinger-metadata-versus-support"

},

{

"canonical_proposition": "DeepResearch Bench evaluated binary support for deduplicated statement-URL pairs; numeric values are edition-specific.",

"claim_occurrence_id": "V2-SOL-03:r3_deepresearch_bench_fact",

"correction_ids": [

"correction:independent-semantic-warrant-gaps"

],

"raw_proposition": "DeepResearch Bench's FACT framework measured whether retrieved page text supported each deduplicated statement-URL pair.",

"review_status": "independent-review-no-credit-insufficient-span-semantics",

"source_citation_occurrence_ids": [

"V2-SOL-03:s2",

"V2-SOL-03:s7"

],

"span_occurrence_ids": [

"V2-SOL-03:s2:s2_span1",

"V2-SOL-03:s7:s7_span1"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deepresearch-bench-fact"

},

{

"canonical_proposition": "DeepTRACE measured statement-source support and found between-system variation; its Gemini value is internally inconsistent between table and prose.",

"claim_occurrence_id": "V2-SOL-03:r4_deeptrace_support",

"correction_ids": [

"correction:deeptrace-gemini-number"

],

"raw_proposition": "DeepTRACE found wide variation in both citation accuracy and the fraction of relevant statements supported by any listed source.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-03:s3"

],

"span_occurrence_ids": [

"V2-SOL-03:s3:s3_span1"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deeptrace-support-variation"

},

{

"canonical_proposition": "The URL-health study measured HTTP resolution and Wayback presence, not semantic claim support, on outputs including reused DeepResearch Bench material.",

"claim_occurrence_id": "V2-SOL-03:r5_deep_agent_url_resolution",

"correction_ids": [],

"raw_proposition": "On pre-collected DeepResearch Bench outputs, both tested commercial deep-research agents produced non-resolving and operationally hallucinated URLs.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-03:s4",

"V2-SOL-03:s2"

],

"span_occurrence_ids": [

"V2-SOL-03:s4:s4_span1",

"V2-SOL-03:s4:s4_span2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:url-health-resolution"

},

{

"canonical_proposition": "The URL-health paper's bounded correction loop reduced non-resolving URLs but did not evaluate surviving claim support.",

"claim_occurrence_id": "V2-SOL-03:r6_url_self_correction",

"correction_ids": [],

"raw_proposition": "An agentic URL-checking loop substantially reduced non-resolving URLs, but did not test whether surviving URLs supported the claims.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-03:s4"

],

"span_occurrence_ids": [

"V2-SOL-03:s4:s4_span3"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:url-health-correction-loop"

},

{

"canonical_proposition": "Cited but Not Verified operationalized link access, topical relevance, and factual support separately and reported materially different rates.",

"claim_occurrence_id": "V2-SOL-03:r7_source_attribution_framework",

"correction_ids": [],

"raw_proposition": "Across model-agent configurations, working links and topical relevance were consistently higher than factual support.",

"review_status": "author-candidate-supported-pending-independent-review",

"source_citation_occurrence_ids": [

"V2-SOL-03:s5"

],

"span_occurrence_ids": [

"V2-SOL-03:s5:s5_span1"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-link-relevance-support-gap"

},

{

"canonical_proposition": "Within the Cited but Not Verified harness, the 2-to-150-call ablation reduced reported Fact Check scores for two setups without establishing a general causal law about search depth.",

"claim_occurrence_id": "V2-SOL-03:r8_search_depth_ablation",

"correction_ids": [

"correction:independent-semantic-warrant-gaps"

],

"raw_proposition": "In one controlled harness, increasing the permitted tool calls left link metrics high but reduced factual-support rates for two models.",

"review_status": "independent-review-no-credit-insufficient-span-semantics",

"source_citation_occurrence_ids": [

"V2-SOL-03:s5"

],

"span_occurrence_ids": [

"V2-SOL-03:s5:s5_span2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-search-depth-ablation"

},

{

"canonical_proposition": "A human-reviewed verifier benchmark showed that rubric-LLM citation-support judgments retain model-dependent error and directional bias.",

"claim_occurrence_id": "V2-SOL-03:r9_verifier_calibration",

"correction_ids": [

"correction:independent-semantic-warrant-gaps"

],

"raw_proposition": "Automated support judgments themselves remain imperfect, including on a fully human-reviewed adversarial citation benchmark.",

"review_status": "independent-review-no-credit-insufficient-span-semantics",

"source_citation_occurrence_ids": [

"V2-SOL-03:s6"

],

"span_occurrence_ids": [

"V2-SOL-03:s6:s6_span1"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:citation-verifier-calibration"

},

{

"canonical_proposition": "Cited but Not Verified operationalized link access, topical relevance, and factual support separately and reported materially different rates.",

"claim_occurrence_id": "V2-SOL-04:R1",

"correction_ids": [

"correction:independent-semantic-warrant-gaps"

],

"raw_proposition": "In the broadest direct end-to-end audit found, resolving and topically relevant citations were substantially more common than citations whose retrieved content supported the attributed claim.",

"review_status": "independent-review-no-credit-insufficient-span-semantics",

"source_citation_occurrence_ids": [

"V2-SOL-04:S1"

],

"span_occurrence_ids": [

"V2-SOL-04:S1:S1-SP1",

"V2-SOL-04:S1:S1-SP2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-link-relevance-support-gap"

},

{

"canonical_proposition": "Within the Cited but Not Verified harness, the 2-to-150-call ablation reduced reported Fact Check scores for two setups without establishing a general causal law about search depth.",

"claim_occurrence_id": "V2-SOL-04:R2",

"correction_ids": [

"correction:independent-semantic-warrant-gaps"

],

"raw_proposition": "Increasing the allowed search depth did not improve claim support in the controlled two-model ablation and coincided with a large decline while access and relevance remained high.",

"review_status": "independent-review-no-credit-insufficient-span-semantics",

"source_citation_occurrence_ids": [

"V2-SOL-04:S1"

],

"span_occurrence_ids": [

"V2-SOL-04:S1:S1-SP3"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-search-depth-ablation"

},

{

"canonical_proposition": "DeepResearch Bench evaluated binary support for deduplicated statement-URL pairs; numeric values are edition-specific.",

"claim_occurrence_id": "V2-SOL-04:R3",

"correction_ids": [

"correction:drbench-edition-drift",

"correction:independent-semantic-warrant-gaps"

],

"raw_proposition": "DeepResearch Bench's FACT evaluator found substantial but imperfect claim support for all four tested commercial deep-research agents.",

"review_status": "independent-review-no-credit-insufficient-span-semantics",

"source_citation_occurrence_ids": [

"V2-SOL-04:S2"

],

"span_occurrence_ids": [

"V2-SOL-04:S2:S2-SP1",

"V2-SOL-04:S2:S2-SP2",

"V2-SOL-04:S2:S2-SP3"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deepresearch-bench-fact"

},

{

"canonical_proposition": "ReportBench compared cited statements with retrieved cited-page content and reported sub-100-percent semantic match rates in its bounded survey-task setting.",

"claim_occurrence_id": "V2-SOL-04:R4",

"correction_ids": [

"correction:independent-semantic-warrant-gaps"

],

"raw_proposition": "ReportBench measured semantic consistency between cited statements and retrieved cited pages, finding imperfect citation match rates for both tested deep-research products.",

"review_status": "independent-review-no-credit-insufficient-span-semantics",

"source_citation_occurrence_ids": [

"V2-SOL-04:S3"

],

"span_occurrence_ids": [

"V2-SOL-04:S3:S3-SP1",

"V2-SOL-04:S3:S3-SP2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:reportbench-match-rate"

},

{

"canonical_proposition": "DeepTRACE measured statement-source support and found between-system variation; its Gemini value is internally inconsistent between table and prose.",

"claim_occurrence_id": "V2-SOL-04:R5",

"correction_ids": [

"correction:deeptrace-gemini-number"

],

"raw_proposition": "DeepTRACE's citation-to-factual-support matrix found wide variation in whether a cited source actually supported the cited statement, and separately found large shares of relevant statements unsupported by any listed source.",

"review_status": "qualified-no-credit-for-corrected-component",

"source_citation_occurrence_ids": [

"V2-SOL-04:S4"

],

"span_occurrence_ids": [

"V2-SOL-04:S4:S4-SP1",

"V2-SOL-04:S4:S4-SP2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deeptrace-support-variation"

},

{

"canonical_proposition": "In one five-system dermatology-review audit, reference identifiability and metadata correctness did not establish sentence-level claim-citation concordance.",

"claim_occurrence_id": "V2-SOL-04:R6",

"correction_ids": [],

"raw_proposition": "A human-authored dermatology audit found most ChatGPT Deep Research and Le Chat references identifiable, but full metadata correctness was lower.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-04:S5",

"V2-SOL-04:S6"

],

"span_occurrence_ids": [

"V2-SOL-04:S5:S5-SP1",

"V2-SOL-04:S5:S5-SP2",

"V2-SOL-04:S6:S6-SP1"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:keplinger-metadata-versus-support"

},

{

"canonical_proposition": "In one five-system dermatology-review audit, reference identifiability and metadata correctness did not establish sentence-level claim-citation concordance.",

"claim_occurrence_id": "V2-SOL-04:R7",

"correction_ids": [],

"raw_proposition": "In the same dermatology audit, citation-bearing sentences had high error rates even though reference identifiability exceeded 90%.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-SOL-04:S5",

"V2-SOL-04:S6"

],

"span_occurrence_ids": [

"V2-SOL-04:S5:S5-SP3",

"V2-SOL-04:S6:S6-SP1"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:keplinger-metadata-versus-support"

},

{

"canonical_proposition": "A human-reviewed verifier benchmark showed that rubric-LLM citation-support judgments retain model-dependent error and directional bias.",

"claim_occurrence_id": "V2-SOL-04:R8",

"correction_ids": [

"correction:independent-semantic-warrant-gaps"

],

"raw_proposition": "A subsequent human-reviewed verifier benchmark found that scalar judge scores can conceal directional bias, qualifying confidence in LLM-judged citation-support rates.",

"review_status": "independent-review-no-credit-insufficient-span-semantics",

"source_citation_occurrence_ids": [

"V2-SOL-04:S7"

],

"span_occurrence_ids": [

"V2-SOL-04:S7:S7-SP1",

"V2-SOL-04:S7:S7-SP2"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:citation-verifier-calibration"

},

{

"canonical_proposition": "DeepResearch Bench evaluated binary support for deduplicated statement-URL pairs; numeric values are edition-specific.",

"claim_occurrence_id": "V2-TERRA-01:r1_deepresearchbench_fact",

"correction_ids": [

"correction:drbench-edition-drift"

],

"raw_proposition": "DeepResearch Bench FACT measured whether extracted, deduplicated statement-URL pairs had webpage evidence sufficient to support the statement.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-01:s1_deepresearchbench"

],

"span_occurrence_ids": [

"V2-TERRA-01:s1_deepresearchbench:s1_method",

"V2-TERRA-01:s1_deepresearchbench:s1_table",

"V2-TERRA-01:s1_deepresearchbench:s1_dates"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deepresearch-bench-fact"

},

{

"canonical_proposition": "ReportBench compared cited statements with retrieved cited-page content and reported sub-100-percent semantic match rates in its bounded survey-task setting.",

"claim_occurrence_id": "V2-TERRA-01:r2_reportbench_cited_statement_match",

"correction_ids": [],

"raw_proposition": "ReportBench measured semantic consistency of cited statements against retrieved content from their cited webpages.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-01:s2_reportbench"

],

"span_occurrence_ids": [

"V2-TERRA-01:s2_reportbench:s2_method",

"V2-TERRA-01:s2_reportbench:s2_metric_table",

"V2-TERRA-01:s2_reportbench:s2_collection"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:reportbench-match-rate"

},

{

"canonical_proposition": "ResearcherBench separated support precision among cited claims from citation coverage across factual claims.",

"claim_occurrence_id": "V2-TERRA-01:r3_researcherbench_faithfulness_and_groundedness",

"correction_ids": [],

"raw_proposition": "ResearcherBench measured whether a cited claim is supported by the textual content extracted from its citation URL, and separately measured citation coverage of all factual claims.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-01:s3_researcherbench"

],

"span_occurrence_ids": [

"V2-TERRA-01:s3_researcherbench:s3_definition",

"V2-TERRA-01:s3_researcherbench:s3_models_time",

"V2-TERRA-01:s3_researcherbench:s3_table"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:researcherbench-faithfulness-groundedness"

},

{

"canonical_proposition": "Cited but Not Verified operationalized link access, topical relevance, and factual support separately and reported materially different rates.",

"claim_occurrence_id": "V2-TERRA-01:r4_cited_not_verified_source_attribution",

"correction_ids": [

"correction:cnv-query-source-scope"

],

"raw_proposition": "Cited but Not Verified jointly evaluated link accessibility, topical relevance, and factual claim support against retrieved cited content.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-01:s4_cited_not_verified"

],

"span_occurrence_ids": [

"V2-TERRA-01:s4_cited_not_verified:s4_definition_table",

"V2-TERRA-01:s4_cited_not_verified:s4_depth"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-link-relevance-support-gap"

},

{

"canonical_proposition": "The URL-health study measured HTTP resolution and Wayback presence, not semantic claim support, on outputs including reused DeepResearch Bench material.",

"claim_occurrence_id": "V2-TERRA-01:r5_urlhealth_resolution",

"correction_ids": [],

"raw_proposition": "Detecting and Correcting Reference Hallucinations measured whether citation URLs resolved and operationally classified non-resolving URLs as archived/stale or absent from the Wayback Machine/hallucinated.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-01:s5_urlhealth"

],

"span_occurrence_ids": [

"V2-TERRA-01:s5_urlhealth:s5_scope",

"V2-TERRA-01:s5_urlhealth:s5_table",

"V2-TERRA-01:s5_urlhealth:s5_comparison"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:url-health-resolution"

},

{

"canonical_proposition": "DeepResearch Bench evaluated binary support for deduplicated statement-URL pairs; numeric values are edition-specific.",

"claim_occurrence_id": "V2-TERRA-02:r1_deepresearchbench_fact",

"correction_ids": [

"correction:drbench-edition-drift"

],

"raw_proposition": "DeepResearch Bench's FACT evaluation found differing citation-support precision and supported-citation volume among four commercial deep-research agents.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-02:s1_deepresearchbench"

],

"span_occurrence_ids": [

"V2-TERRA-02:s1_deepresearchbench:s1_fact_method",

"V2-TERRA-02:s1_deepresearchbench:s1_fact_results",

"V2-TERRA-02:s1_deepresearchbench:s1_fact_formula",

"V2-TERRA-02:s1_deepresearchbench:s1_fact_judge_validation"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deepresearch-bench-fact"

},

{

"canonical_proposition": "LiveResearchBench v5 separately typed inaccessible/non-resolving, irrelevant, and unsupported-claim citation errors for two task categories; its table reports per-report errors without raw denominators.",

"claim_occurrence_id": "V2-TERRA-02:r2_liveresearchbench_wide_info",

"correction_ids": [],

"raw_proposition": "LiveResearchBench found non-zero citation errors for three leading systems on its Wide Info Search tasks, with unsupported claims the largest reported error class for each system.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-02:s2_liveresearchbench"

],

"span_occurrence_ids": [

"V2-TERRA-02:s2_liveresearchbench:s2_rubric_tree",

"V2-TERRA-02:s2_liveresearchbench:s2_table7"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:liveresearchbench-e1-e2-e3"

},

{

"canonical_proposition": "LiveResearchBench v5 separately typed inaccessible/non-resolving, irrelevant, and unsupported-claim citation errors for two task categories; its table reports per-report errors without raw denominators.",

"claim_occurrence_id": "V2-TERRA-02:r3_liveresearchbench_market_analysis",

"correction_ids": [],

"raw_proposition": "The same LiveResearchBench evaluation found higher average citation-error totals on Market Analysis than Wide Info Search for all three evaluated systems.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-02:s2_liveresearchbench"

],

"span_occurrence_ids": [

"V2-TERRA-02:s2_liveresearchbench:s2_table7"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:liveresearchbench-e1-e2-e3"

},

{

"canonical_proposition": "DeepTRACE measured statement-source support and found between-system variation; its Gemini value is internally inconsistent between table and prose.",

"claim_occurrence_id": "V2-TERRA-02:r4_deeptrace",

"correction_ids": [

"correction:deeptrace-gemini-number"

],

"raw_proposition": "DeepTRACE measured citation accuracy and unsupported-statement rates for public deep-research configurations, finding substantial variation rather than universal success or failure.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-02:s3_deeptrace"

],

"span_occurrence_ids": [

"V2-TERRA-02:s3_deeptrace:s3_definition",

"V2-TERRA-02:s3_deeptrace:s3_table1",

"V2-TERRA-02:s3_deeptrace:s3_corpus_and_retrieval",

"V2-TERRA-02:s3_deeptrace:s3_judge_validation"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deeptrace-support-variation"

},

{

"canonical_proposition": "ResearcherBench separated support precision among cited claims from citation coverage across factual claims.",

"claim_occurrence_id": "V2-TERRA-02:r5_researcherbench",

"correction_ids": [],

"raw_proposition": "ResearcherBench's factual-assessment pipeline found high cited-claim support precision but substantially lower citation coverage for several deep research systems on frontier-AI questions.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-02:s4_researcherbench"

],

"span_occurrence_ids": [

"V2-TERRA-02:s4_researcherbench:s4_method",

"V2-TERRA-02:s4_researcherbench:s4_results"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:researcherbench-faithfulness-groundedness"

},

{

"canonical_proposition": "Cited but Not Verified operationalized link access, topical relevance, and factual support separately and reported materially different rates.",

"claim_occurrence_id": "V2-TERRA-03:R1",

"correction_ids": [],

"raw_proposition": "In a 14-model web-search research-agent experiment, working URLs and topical relevance were much higher than factual claim support.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-03:S1"

],

"span_occurrence_ids": [

"V2-TERRA-03:S1:S1-A",

"V2-TERRA-03:S1:S1-B",

"V2-TERRA-03:S1:S1-C"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-link-relevance-support-gap"

},

{

"canonical_proposition": "Within the Cited but Not Verified harness, the 2-to-150-call ablation reduced reported Fact Check scores for two setups without establishing a general causal law about search depth.",

"claim_occurrence_id": "V2-TERRA-03:R2",

"correction_ids": [],

"raw_proposition": "Increasing permitted search depth reduced the reported factual-support score in a controlled two-model ablation while link accessibility and relevance stayed high.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-03:S1"

],

"span_occurrence_ids": [

"V2-TERRA-03:S1:S1-D",

"V2-TERRA-03:S1:S1-E"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-search-depth-ablation"

},

{

"canonical_proposition": "DeepResearch Bench evaluated binary support for deduplicated statement-URL pairs; numeric values are edition-specific.",

"claim_occurrence_id": "V2-TERRA-03:R3",

"correction_ids": [

"correction:drbench-edition-drift"

],

"raw_proposition": "DeepResearch Bench reported statement-URL support precision for four commercial deep-research agents on 100 expert-designed tasks.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-03:S2",

"V2-TERRA-03:S5"

],

"span_occurrence_ids": [

"V2-TERRA-03:S2:S2-A",

"V2-TERRA-03:S2:S2-B",

"V2-TERRA-03:S2:S2-C",

"V2-TERRA-03:S2:S2-D",

"V2-TERRA-03:S5:S5-A",

"V2-TERRA-03:S5:S5-B"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deepresearch-bench-fact"

},

{

"canonical_proposition": "ReportBench compared cited statements with retrieved cited-page content and reported sub-100-percent semantic match rates in its bounded survey-task setting.",

"claim_occurrence_id": "V2-TERRA-03:R4",

"correction_ids": [],

"raw_proposition": "ReportBench reported semantic consistency between cited statements and retrieved cited webpages for two commercial deep-research products on academic-survey tasks.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-03:S3"

],

"span_occurrence_ids": [

"V2-TERRA-03:S3:S3-A",

"V2-TERRA-03:S3:S3-B",

"V2-TERRA-03:S3:S3-C",

"V2-TERRA-03:S3:S3-D"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:reportbench-match-rate"

},

{

"canonical_proposition": "In one five-system dermatology-review audit, reference identifiability and metadata correctness did not establish sentence-level claim-citation concordance.",

"claim_occurrence_id": "V2-TERRA-03:R5",

"correction_ids": [],

"raw_proposition": "A preliminary biomedical evaluation checked bibliographic identifiability and citation-bearing-sentence concordance in five research systems and found both fabricated references and source-claim inaccuracies.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-03:S4",

"V2-TERRA-03:S6"

],

"span_occurrence_ids": [

"V2-TERRA-03:S4:S4-A",

"V2-TERRA-03:S4:S4-B",

"V2-TERRA-03:S4:S4-C",

"V2-TERRA-03:S6:S6-A"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:keplinger-metadata-versus-support"

},

{

"canonical_proposition": "DeepResearch Bench evaluated binary support for deduplicated statement-URL pairs; numeric values are edition-specific.",

"claim_occurrence_id": "V2-TERRA-04:R1_deepresearchbench_support",

"correction_ids": [],

"raw_proposition": "In DeepResearch Bench's FACT evaluation, a cited statement-URL pair was counted as accurate only when retrieved page text supported the claim; citation accuracy varied across four commercial deep-research agents.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-04:S1_deepresearchbench"

],

"span_occurrence_ids": [

"V2-TERRA-04:S1_deepresearchbench:S1_method",

"V2-TERRA-04:S1_deepresearchbench:S1_table1",

"V2-TERRA-04:S1_deepresearchbench:S1_metric"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:deepresearch-bench-fact"

},

{

"canonical_proposition": "ResearcherBench separated support precision among cited claims from citation coverage across factual claims.",

"claim_occurrence_id": "V2-TERRA-04:R2_researcherbench_support",

"correction_ids": [],

"raw_proposition": "ResearcherBench measured whether cited claims were supported by text extracted from their cited URLs, reporting high but non-perfect faithfulness alongside much lower citation coverage.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-04:S2_researcherbench"

],

"span_occurrence_ids": [

"V2-TERRA-04:S2_researcherbench:S2_method",

"V2-TERRA-04:S2_researcherbench:S2_table2",

"V2-TERRA-04:S2_researcherbench:S2_scope"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:researcherbench-faithfulness-groundedness"

},

{

"canonical_proposition": "ReportBench compared cited statements with retrieved cited-page content and reported sub-100-percent semantic match rates in its bounded survey-task setting.",

"claim_occurrence_id": "V2-TERRA-04:R3_reportbench_support",

"correction_ids": [],

"raw_proposition": "ReportBench's cited-statement pipeline retrieved cited webpages, located a semantically relevant passage, and checked consistency; OpenAI Deep Research and Gemini Deep Research had reported cited-statement match rates below 100%.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-04:S3_reportbench"

],

"span_occurrence_ids": [

"V2-TERRA-04:S3_reportbench:S3_method",

"V2-TERRA-04:S3_reportbench:S3_table1",

"V2-TERRA-04:S3_reportbench:S3_time"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:reportbench-match-rate"

},

{

"canonical_proposition": "The URL-health study measured HTTP resolution and Wayback presence, not semantic claim support, on outputs including reused DeepResearch Bench material.",

"claim_occurrence_id": "V2-TERRA-04:R4_urlhealth_resolution",

"correction_ids": [],

"raw_proposition": "A large-scale URL-liveness study directly measured non-resolving and likely-nonexistent citation URLs from two deep-research agents in DRBench.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-04:S4_urlhealth"

],

"span_occurrence_ids": [

"V2-TERRA-04:S4_urlhealth:S4_table2",

"V2-TERRA-04:S4_urlhealth:S4_comparison",

"V2-TERRA-04:S4_urlhealth:S4_method"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:url-health-resolution"

},

{

"canonical_proposition": "Cited but Not Verified operationalized link access, topical relevance, and factual support separately and reported materially different rates.",

"claim_occurrence_id": "V2-TERRA-04:R5_cited_not_verified_support_and_resolution",

"correction_ids": [],

"raw_proposition": "A later end-to-end study of citation-claim pairs found high working-link and topical-relevance rates but materially lower Fact Check support rates across 14 web-search research-agent configurations.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-04:S5_cited_not_verified"

],

"span_occurrence_ids": [

"V2-TERRA-04:S5_cited_not_verified:S5_method",

"V2-TERRA-04:S5_cited_not_verified:S5_table1",

"V2-TERRA-04:S5_cited_not_verified:S5_scope"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-link-relevance-support-gap"

},

{

"canonical_proposition": "Within the Cited but Not Verified harness, the 2-to-150-call ablation reduced reported Fact Check scores for two setups without establishing a general causal law about search depth.",

"claim_occurrence_id": "V2-TERRA-04:R6_search_depth_ablation",

"correction_ids": [],

"raw_proposition": "In the same later study, increasing allowed search-tool calls lowered Fact Check support while Link Works and relevance stayed high for two frontier model-agent configurations.",

"review_status": "unresolved-no-warrant-credit",

"source_citation_occurrence_ids": [

"V2-TERRA-04:S5_cited_not_verified"

],

"span_occurrence_ids": [

"V2-TERRA-04:S5_cited_not_verified:S5_depth"

],

"warrant_dimensions": {

"causality": "no causal interpretation unless the bounded source experiment itself supports it",

"comparison_class": "compare only rows sharing a published table and method",

"metric_definition": "do not pool URL liveness, relevance, cited-claim support, and claim coverage",

"modality": "retain source-reported or observational modality; no universalization",

"numerical_strengthening": "retain published precision and unknown denominators; edition conflicts receive no numeric credit",

"population": "bind to the published task or report population",

"scope": "bind to the named benchmark, systems, prompt, and retrieval pipeline",

"source_versus_agent_wording": "the canonical proposition is narrower than the raw answer where a correction or unknown applies",

"time": "bind to the named source edition and dated product snapshot"

},

"warrant_id": "warrant:cnv-search-depth-ablation"

}

],

"corrections": [

{

"correction_id": "correction:liveresearchbench-license",

"effect": "license metadata corrected; scientific warrant unchanged",

"evidence": "The arXiv v5 abstract page links http://creativecommons.org/licenses/by-nc-sa/4.0/.",

"kind": "license",

"raw_occurrences": [

"V2-TERRA-02:s2_liveresearchbench"

],

"raw_value": "CC BY 4.0 for arXiv paper",

"review_value": "CC BY-NC-SA 4.0"

},

{

"correction_id": "correction:cnv-query-source-scope",

"effect": "raw scope receives no credit; canonical warrant retains both upstream roots",

"evidence": "Cited but Not Verified section 3.4 states both upstream query sources.",

"kind": "scope",

"raw_occurrences": [

"V2-TERRA-01:r4_cited_not_verified_source_attribution"

],

"raw_value": "130 research queries drawn from DeepResearch Bench",

"review_value": "130 research queries drawn from DeepResearch Bench and BrowseComp"

},

{

"correction_id": "correction:deeptrace-gemini-number",

"effect": "generic between-system variation remains; exact Gemini numeric warrant is withheld",

"evidence": "DeepTRACE ICLR 2026 PDF p.8 Table 1 and adjacent results prose disagree.",

"kind": "internal-source-conflict",

"raw_occurrences": [

"V2-SOL-01:r4_deeptrace_support",

"V2-SOL-03:r4_deeptrace_support",

"V2-SOL-04:R5",

"V2-TERRA-02:r4_deeptrace"

],

"raw_value": "Gemini Deep Research citation accuracy 50.3% from Table 1",

"review_value": "unresolved: Table 1 reports 50.3%, surrounding prose reports 40.3%"

},

{

"correction_id": "correction:drbench-edition-drift",

"effect": "method-level warrant remains; numeric roots are edition-specific and never pooled",

"evidence": "v1 reports 90.24/81.44/77.96 for Perplexity/Gemini/OpenAI; ICLR reports 82.63/78.30/75.01.",

"kind": "edition",

"raw_occurrences": [

"V2-SOL-01:r3_deepresearch_bench_fact",

"V2-SOL-02:R3_deepresearch_bench_fact",

"V2-SOL-04:R3",

"V2-TERRA-01:r1_deepresearchbench_fact",

"V2-TERRA-02:r1_deepresearchbench_fact",

"V2-TERRA-03:R3"

],

"raw_value": "arXiv v1 and ICLR values appear across the run matrix",

"review_value": "arXiv v1 and ICLR 2026 are separate editions with materially different FACT values"

},

{

"correction_id": "correction:four-independent-benchmarks",

"effect": "raw independence claim receives no credit",

"evidence": "The ledger records upstream DRBench reuse plus shared Jina Reader and judge-method roots.",

"kind": "dependence",

"raw_occurrences": [

"V2-TERRA-02:answer"

],

"raw_value": "Four independent benchmark papers",

"review_value": "separate papers with known shared task, retrieval, and LLM-judge dependencies"

},

{

"correction_id": "correction:mendeley-file-readback",

"effect": "landing-page metadata can support artifact identity; file-content spans receive no credit",

"evidence": "Fresh Mendeley landing-page reads returned 200; api.mendeley.com/datasets/3s73z9zf3c/versions/2 returned 401.",

"kind": "access",

"raw_occurrences": [

"V2-SOL-01:s1a_keplinger_data",

"V2-SOL-02:S3",

"V2-SOL-04:S6"

],

"raw_value": "supplementary files publicly retrieved or downloadable",

"review_value": "landing pages retrieved; credential-free API returned 401 and file bytes were not captured in this review"

},

{

"correction_id": "correction:independent-semantic-warrant-gaps",

"effect": "the complete raw proposition receives no credit; raw capture and narrower matched fragments remain visible",

"evidence": "Independent exact-edition review completed all 72 credited span roots and found these nine claim occurrences exceed the semantics of their linked spans.",

"kind": "semantic-warrant",

"raw_occurrences": [

"V2-SOL-02:R5_deeptrace_audit",

"V2-SOL-03:r3_deepresearch_bench_fact",

"V2-SOL-03:r8_search_depth_ablation",

"V2-SOL-03:r9_verifier_calibration",

"V2-SOL-04:R1",

"V2-SOL-04:R2",

"V2-SOL-04:R3",

"V2-SOL-04:R4",

"V2-SOL-04:R8"

],

"raw_value": "author-candidate support for the complete raw proposition",

"review_value": "linked quote fragments establish only a subset of the asserted method, comparison, metric, scope, or directional result"

}

],

"count_grammar": {

"candidate_warrant_root": "one reviewed proposition plus data, method, and derivation lineage with at least one claim whose matched spans semantically support the canonical proposition; it remains candidate until independent exact-head review",

"cited_url": "one distinct requested URL string in the eight raw answers",

"exact_span_root": "one edition, locator, and quote identity with an independently captured normalized or ordered-fragment match",

"examined_edition_root": "one versioned content state; mirrors and carriers do not create another edition",

"invalid_citation": "a malformed URL, a URL resolving to a different work, or a source identity contradicted by authoritative metadata",

"report": "one terminal answer artifact bound to one frozen v2 run slot",

"resolving_url_root": "one distinct requested URL whose fresh readback returned usable public source content",

"source_work_root": "one intellectual or data work after URL-carrier normalization",

"unresolved_citation": "one source occurrence with an inaccessible carrier, no independently matched span, or an unresolved identity or warrant correction"

},

"counts": {

"candidate_warrant_roots": 7,

"captured_reports": 8,

"citation_occurrences": 48,

"cited_urls": 30,

"exact_span_roots": 72,

"examined_edition_roots": 14,

"inaccessible_citations": 3,

"independently_confirmed_warrant_roots": 0,

"invalid_citations": 0,

"non_resolving_citations": 0,

"raw_claim_occurrences": 52,

"raw_span_occurrences": 127,

"resolving_url_roots": 27,

"source_work_roots": 11,

"unresolved_citations": 34,

"unsupported_or_force_raised_claims": 20

},

"dependence_edges": [

{

"dimension": "upstream_citation",

"from": "work:cited-not-verified",

"note": "The paper states that its 130 queries are drawn from DeepResearch Bench and BrowseComp.",

"status": "known",

"to": "work:deepresearch-bench-paper"

},

{

"dimension": "upstream_citation",

"from": "work:url-health",

"note": "The URL-health study reuses DeepResearch Bench outputs for part of its corpus.",

"status": "known",

"to": "work:deepresearch-bench-paper"

},

{

"dimension": "data",

"from": "work:keplinger-supplement",

"note": "The supplement contains prompts, outputs, and evaluation material for the article.",

"status": "same-data-root",

"to": "work:keplinger-dermatology-audit"

},

{

"dimension": "data",

"from": "work:deepresearch-bench-repository",

"note": "The repository is the paper's official task/evaluator artifact, not independent evidence.",

"status": "same-data-root",

"to": "work:deepresearch-bench-paper"

},

{

"dimension": "method",

"from": "method:fact-statement-url-support-evaluator",

"status": "known",

"to": "method:llm-judge-after-web-extraction"

},

{

"dimension": "method",

"from": "method:deeptrace-statement-source-matrix",

"status": "known",

"to": "method:llm-judge-after-web-extraction"

},

{

"dimension": "method",

"from": "method:cnv-ast-access-relevance-fact-check",

"status": "known",

"to": "method:llm-judge-after-web-extraction"

},

{

"dimension": "method",

"from": "method:reportbench-statement-source-semantic-match",

"status": "known",

"to": "method:llm-judge-after-web-extraction"

},

{

"dimension": "method",

"from": "method:researcherbench-faithfulness-groundedness",

"status": "known",

"to": "method:llm-judge-after-web-extraction"

},

{

"dimension": "retrieval_infrastructure",

"from": "method:fact-statement-url-support-evaluator",

"status": "known",

"to": "retrieval:jina-reader"

},

{

"dimension": "retrieval_infrastructure",

"from": "method:deeptrace-statement-source-matrix",

"status": "known",

"to": "retrieval:jina-reader"

},

{

"dimension": "retrieval_infrastructure",

"from": "method:researcherbench-faithfulness-groundedness",

"status": "known",

"to": "retrieval:jina-reader"

},

{

"dimension": "data",

"from": "warrant:keplinger-metadata-versus-support",

"status": "declared-reviewed-lineage",

"to": "data:keplinger-three-run-five-system-audit"

},

{

"dimension": "method",

"from": "warrant:keplinger-metadata-versus-support",

"status": "declared-reviewed-lineage",

"to": "method:human-reference-and-sentence-concordance-audit"

},

{

"dimension": "derivation",

"from": "warrant:keplinger-metadata-versus-support",

"status": "declared-reviewed-lineage",

"to": "derivation:keplinger-published-tables-and-figure"

},

{

"dimension": "data",

"from": "warrant:deepresearch-bench-fact",

"status": "declared-reviewed-lineage",

"to": "data:deepresearch-bench-100-tasks-and-system-outputs"

},

{

"dimension": "method",

"from": "warrant:deepresearch-bench-fact",

"status": "declared-reviewed-lineage",

"to": "method:fact-statement-url-support-evaluator"

},

{

"dimension": "derivation",

"from": "warrant:deepresearch-bench-fact",

"status": "declared-reviewed-lineage",

"to": "derivation:deepresearch-bench-edition-specific-tables"

},

{

"dimension": "data",

"from": "warrant:deeptrace-support-variation",

"status": "declared-reviewed-lineage",

"to": "data:deeptrace-303-questions-2727-outputs"

},

{

"dimension": "method",

"from": "warrant:deeptrace-support-variation",

"status": "declared-reviewed-lineage",

"to": "method:deeptrace-statement-source-matrix"

},

{

"dimension": "derivation",

"from": "warrant:deeptrace-support-variation",

"status": "declared-reviewed-lineage",

"to": "derivation:deeptrace-tables-and-prose"

},

{

"dimension": "data",

"from": "warrant:cnv-link-relevance-support-gap",

"status": "declared-reviewed-lineage",

"to": "data:cnv-130-drbench-and-browsecomp-queries"

},

{

"dimension": "method",

"from": "warrant:cnv-link-relevance-support-gap",

"status": "declared-reviewed-lineage",

"to": "method:cnv-ast-access-relevance-fact-check"

},

{

"dimension": "derivation",

"from": "warrant:cnv-link-relevance-support-gap",

"status": "declared-reviewed-lineage",

"to": "derivation:cnv-main-table"

},

{

"dimension": "data",

"from": "warrant:cnv-search-depth-ablation",

"status": "declared-reviewed-lineage",

"to": "data:cnv-depth-ablation-two-model-setups"

},

{

"dimension": "method",

"from": "warrant:cnv-search-depth-ablation",

"status": "declared-reviewed-lineage",

"to": "method:cnv-ast-access-relevance-fact-check"

},

{

"dimension": "derivation",

"from": "warrant:cnv-search-depth-ablation",

"status": "declared-reviewed-lineage",

"to": "derivation:cnv-depth-ablation-tables"

},

{

"dimension": "data",

"from": "warrant:url-health-resolution",

"status": "declared-reviewed-lineage",

"to": "data:url-health-drbench-and-expertqa-urls"

},

{

"dimension": "method",

"from": "warrant:url-health-resolution",

"status": "declared-reviewed-lineage",

"to": "method:http-resolution-wayback-classification"

},

{

"dimension": "derivation",

"from": "warrant:url-health-resolution",

"status": "declared-reviewed-lineage",

"to": "derivation:url-health-observational-results"

},

{

"dimension": "data",

"from": "warrant:url-health-correction-loop",

"status": "declared-reviewed-lineage",

"to": "data:url-health-correction-experiment"

},

{

"dimension": "method",

"from": "warrant:url-health-correction-loop",

"status": "declared-reviewed-lineage",

"to": "method:url-checking-correction-loop"

},

{

"dimension": "derivation",

"from": "warrant:url-health-correction-loop",

"status": "declared-reviewed-lineage",

"to": "derivation:url-health-correction-results"

},

{

"dimension": "data",

"from": "warrant:reportbench-match-rate",

"status": "declared-reviewed-lineage",

"to": "data:reportbench-100-survey-tasks"

},

{

"dimension": "method",

"from": "warrant:reportbench-match-rate",

"status": "declared-reviewed-lineage",

"to": "method:reportbench-statement-source-semantic-match"

},

{

"dimension": "derivation",

"from": "warrant:reportbench-match-rate",

"status": "declared-reviewed-lineage",

"to": "derivation:reportbench-table-1"

},

{

"dimension": "data",

"from": "warrant:researcherbench-faithfulness-groundedness",

"status": "declared-reviewed-lineage",

"to": "data:researcherbench-65-frontier-ai-questions"

},

{

"dimension": "method",

"from": "warrant:researcherbench-faithfulness-groundedness",

"status": "declared-reviewed-lineage",

"to": "method:researcherbench-faithfulness-groundedness"

},

{

"dimension": "derivation",

"from": "warrant:researcherbench-faithfulness-groundedness",

"status": "declared-reviewed-lineage",

"to": "derivation:researcherbench-table-2"

},

{

"dimension": "data",

"from": "warrant:liveresearchbench-e1-e2-e3",

"status": "declared-reviewed-lineage",

"to": "data:liveresearchbench-wide-info-and-market-analysis"

},

{

"dimension": "method",

"from": "warrant:liveresearchbench-e1-e2-e3",

"status": "declared-reviewed-lineage",

"to": "method:liveresearchbench-e1-e2-e3-rubric-tree"

},

{

"dimension": "derivation",

"from": "warrant:liveresearchbench-e1-e2-e3",

"status": "declared-reviewed-lineage",

"to": "derivation:liveresearchbench-v5-table-7"

},

{

"dimension": "data",

"from": "warrant:citation-verifier-calibration",

"status": "declared-reviewed-lineage",

"to": "data:citationverbench-1248-human-reviewed-decisions"

},

{

"dimension": "method",

"from": "warrant:citation-verifier-calibration",

"status": "declared-reviewed-lineage",

"to": "method:citationverbench-rubric-judge-comparison"

},

{

"dimension": "derivation",

"from": "warrant:citation-verifier-calibration",

"status": "declared-reviewed-lineage",

"to": "derivation:citationverbench-reported-calibration"

},

{

"dimension": "model_profile",

"from": "report:V2-SOL-01",

"status": "requested-not-independent",

"to": "model-profile:gpt-5.6-sol"

},

{

"dimension": "prompt",

"from": "report:V2-SOL-01",

"status": "shared-exact-bytes",

"to": "prompt:sha256:d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670"

},

{

"dimension": "retrieval_infrastructure",

"from": "report:V2-SOL-01",

"status": "unknown-or-shared",

"to": "retrieval:codex-public-web-implementation-unknown"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s1_keplinger",

"status": "unresolved",

"to": "span:sha256:c29518951f6335c3c613ebed9c4babc2bc1903426992cd90118a999be69b5cdd"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s1_keplinger",

"status": "exact-normalized-match",

"to": "span:sha256:b903dc2284e1b4d80bb6ef948d79bbdbc7d7447e871281b8577284883f75ccfa"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s1_keplinger",

"status": "unresolved",

"to": "span:sha256:0bea3f00dfce1763dae86c14901bee21053f2d8ceee7f9dde4ab33ec78d3eb50"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-01:s1_keplinger",

"status": "retrieved",

"to": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-01:s1_keplinger",

"status": "normalized",

"to": "work:keplinger-dermatology-audit"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-01:s1_keplinger",

"status": "examined",

"to": "edition:keplinger-vor-2025"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s1a_keplinger_data",

"status": "exact-normalized-match",

"to": "span:sha256:6f79c3c824ccf889d73e234335661e5fca68d90f8c78ee97aef488e4fdb32a49"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-01:s1a_keplinger_data",

"status": "retrieved",

"to": "https://data.mendeley.com/datasets/3s73z9zf3c/2"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-01:s1a_keplinger_data",

"status": "normalized",

"to": "work:keplinger-supplement"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-01:s1a_keplinger_data",

"status": "examined",

"to": "edition:keplinger-supplement-v2"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s2_drbench",

"status": "exact-normalized-match",

"to": "span:sha256:7d7331942f9d0520e18e96206e7718e5187788be4d48cef995fac79b235db757"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s2_drbench",

"status": "unresolved",

"to": "span:sha256:6070d8d9e7ba0d923aca7ae3517013c0e865119cea05c3e9cb0565d273eee0ee"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s2_drbench",

"status": "exact-normalized-match",

"to": "span:sha256:c27aaca02bda06ed764be53351158fc862af9d4a556d3e82074507436f48c3f8"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s2_drbench",

"status": "exact-normalized-match",

"to": "span:sha256:279db725380c133044944dccb379b75a396242219aa30437b26ddc40b09a3b5c"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-01:s2_drbench",

"status": "retrieved",

"to": "https://proceedings.iclr.cc/paper_files/paper/2026/file/465f22be10e07b301c6ed58f0472f704-Paper-Conference.pdf"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-01:s2_drbench",

"status": "normalized",

"to": "work:deepresearch-bench-paper"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-01:s2_drbench",

"status": "examined",

"to": "edition:drbench-iclr-2026"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s2a_drbench_repo",

"status": "exact-normalized-match",

"to": "span:sha256:0abbab3d6269d58eb76d8e90eff1e4c27512cf46d9b8e5ff0a9f537b18f232fa"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-01:s2a_drbench_repo",

"status": "retrieved",

"to": "https://github.com/Ayanami0730/deep_research_bench"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-01:s2a_drbench_repo",

"status": "normalized",

"to": "work:deepresearch-bench-repository"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-01:s2a_drbench_repo",

"status": "examined",

"to": "edition:drbench-repo-main-469cce5"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s3_deeptrace",

"status": "unresolved",

"to": "span:sha256:b45b782077327a7123cff72789773ad2f70f5608c8e0e33b0d211b3d60e0de0a"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s3_deeptrace",

"status": "unresolved",

"to": "span:sha256:4e55438723cddd7dfbd56c115e5fe6e809c0091ec7631928d67490aafc49f118"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s3_deeptrace",

"status": "exact-normalized-match",

"to": "span:sha256:a15ab02e9fcb23df02fb03d28967d4220ddf65ad380d223049f22b6ab68ed5eb"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s3_deeptrace",

"status": "exact-normalized-match",

"to": "span:sha256:35927851edd2e7e5e0f60d98498940f88304ba99bf1f85a08535663943c2f40d"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-01:s3_deeptrace",

"status": "retrieved",

"to": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-01:s3_deeptrace",

"status": "normalized",

"to": "work:deeptrace"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-01:s3_deeptrace",

"status": "examined",

"to": "edition:deeptrace-iclr-2026"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s4_cited_not_verified",

"status": "unresolved",

"to": "span:sha256:9339db261929e3b56820214b791131dcbe342e20111e4b08916cfa523893810f"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s4_cited_not_verified",

"status": "unresolved",

"to": "span:sha256:419fcae33b7758515a2334e64b7dca713a139a11d6930bc9cc772559989ae12e"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s4_cited_not_verified",

"status": "exact-normalized-match",

"to": "span:sha256:c3dfc98566c27561eb7267972e00e3e1ba6a399a3ff5d002dca2d1233836790c"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-01:s4_cited_not_verified",

"status": "retrieved",

"to": "https://arxiv.org/abs/2605.06635"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-01:s4_cited_not_verified",

"status": "normalized",

"to": "work:cited-not-verified"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-01:s4_cited_not_verified",

"status": "examined",

"to": "edition:cnv-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s5_url_health",

"status": "exact-normalized-match",

"to": "span:sha256:a18e778fac3d49e81167f05e09fbc361e91af8c0c01b0fec52991f0a1e3d0b91"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s5_url_health",

"status": "unresolved",

"to": "span:sha256:db9dc28183620369f5ad1fb0f179e8aadbc14c69c0fdb57b8a8b79cd8e95e2a6"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-01:s5_url_health",

"status": "unresolved",

"to": "span:sha256:541aadd42c71d1bd9c2c458c8e250d5000ffb75f3801eea832a0a0ef58a3914b"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-01:s5_url_health",

"status": "retrieved",

"to": "https://arxiv.org/abs/2604.03173"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-01:s5_url_health",

"status": "normalized",

"to": "work:url-health"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-01:s5_url_health",

"status": "examined",

"to": "edition:url-health-arxiv-v1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-01:r1_human_dermatology_chatgpt",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-01:s1_keplinger"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-01:r1_human_dermatology_chatgpt",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-01:s1a_keplinger_data"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-01:r2_human_dermatology_lechat",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-01:s1_keplinger"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-01:r2_human_dermatology_lechat",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-01:s1a_keplinger_data"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-01:r3_deepresearch_bench_fact",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-01:s2_drbench"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-01:r3_deepresearch_bench_fact",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-01:s2a_drbench_repo"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-01:r4_deeptrace_support",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-01:s3_deeptrace"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-01:r5_cited_but_not_verified_cross_model",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-01:s4_cited_not_verified"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-01:r6_cited_but_not_verified_depth_ablation",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-01:s4_cited_not_verified"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-01:r7_url_resolution_drbench",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-01:s5_url_health"

},

{

"dimension": "model_profile",

"from": "report:V2-SOL-02",

"status": "requested-not-independent",

"to": "model-profile:gpt-5.6-sol"

},

{

"dimension": "prompt",

"from": "report:V2-SOL-02",

"status": "shared-exact-bytes",

"to": "prompt:sha256:d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670"

},

{

"dimension": "retrieval_infrastructure",

"from": "report:V2-SOL-02",

"status": "unknown-or-shared",

"to": "retrieval:codex-public-web-implementation-unknown"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S1",

"status": "unresolved",

"to": "span:sha256:a7e8b6b9e8aa380a3d1eaedaae1439792bc858a48735284ecbd2822da1c3c0fe"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S1",

"status": "unresolved",

"to": "span:sha256:388e78d0c793138d79699c82540cb28cbd5722adfe6df119b36ee40a4502d9b5"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-02:S1",

"status": "retrieved",

"to": "https://arxiv.org/html/2604.03173v1"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-02:S1",

"status": "normalized",

"to": "work:url-health"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-02:S1",

"status": "examined",

"to": "edition:url-health-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S2",

"status": "exact-normalized-match",

"to": "span:sha256:092745c7db6407d4d52927642741cde229fad05a5347bb1867a5154421060e70"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-02:S2",

"status": "retrieved",

"to": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-02:S2",

"status": "normalized",

"to": "work:keplinger-dermatology-audit"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-02:S2",

"status": "examined",

"to": "edition:keplinger-vor-2025"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S3",

"status": "unresolved",

"to": "span:sha256:870d5c8e029e0c3d25b0a4682848e6caafc22ee35db116517a55bf334eea3e2d"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S3",

"status": "unresolved",

"to": "span:sha256:3bbe9293af4ece39817697cf29b0fe4ab9423d429d9867dec2540b72c82fd5ba"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-02:S3",

"status": "retrieved",

"to": "https://data.mendeley.com/datasets/3s73z9zf3c/2"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-02:S3",

"status": "normalized",

"to": "work:keplinger-supplement"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-02:S3",

"status": "examined",

"to": "edition:keplinger-supplement-v2"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S4",

"status": "unresolved",

"to": "span:sha256:c8683e7dad85f22158370acbb6a6e949b304b41bcba2c06f55cd31e90ce27130"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S4",

"status": "unresolved",

"to": "span:sha256:5899663b10cbe488c635c75743b9cf4ba1aab8c3e3a67604d4fa29248b387324"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-02:S4",

"status": "inaccessible",

"to": "https://openreview.net/pdf?id=hQ0K2Hhq7H"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-02:S4",

"status": "normalized",

"to": "work:deepresearch-bench-paper"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-02:S4",

"status": "examined",

"to": "edition:drbench-iclr-2026"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S5",

"status": "ordered-fragment-match",

"to": "span:sha256:e223640cb0ed12c87ca1d2406f3276f30a3b8d2017dd4dc1457fe59a94471040"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S5",

"status": "unresolved",

"to": "span:sha256:3b6f33fc8d800c8712c7b9fc2daafa5027f5176b23aa9120eeea323a21790a3a"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-02:S5",

"status": "retrieved",

"to": "https://arxiv.org/html/2508.15804v1"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-02:S5",

"status": "normalized",

"to": "work:reportbench"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-02:S5",

"status": "examined",

"to": "edition:reportbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S6",

"status": "exact-normalized-match",

"to": "span:sha256:a08468cb62f2feb21d9cbb95175ce00f54592bd83573cc9166e2375947f62fee"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S6",

"status": "exact-normalized-match",

"to": "span:sha256:e2c9f6220843a04665f5d1cd142ac15f04211ace5c579fa67f77e74e84a76dfe"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-02:S6",

"status": "retrieved",

"to": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/ad08767706825033b99122332293033d-Abstract-Conference.html"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-02:S6",

"status": "normalized",

"to": "work:deeptrace"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-02:S6",

"status": "examined",

"to": "edition:deeptrace-iclr-2026"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S7",

"status": "exact-normalized-match",

"to": "span:sha256:912c931063109f84ea88aea34891dd5e0f9147d2176af258059246431e9abfb1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-02:S7",

"status": "unresolved",

"to": "span:sha256:1f0504defcd0c58184e6958c7cf30ddd2120357feb41c2753301a5912457fef9"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-02:S7",

"status": "retrieved",

"to": "https://arxiv.org/html/2605.06635v1"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-02:S7",

"status": "normalized",

"to": "work:cited-not-verified"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-02:S7",

"status": "examined",

"to": "edition:cnv-arxiv-v1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-02:R1_url_resolution",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-02:S1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-02:R2_human_dermatology_claim_support",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-02:S2"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-02:R2_human_dermatology_claim_support",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-02:S3"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-02:R3_deepresearch_bench_fact",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-02:S4"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-02:R4_reportbench_match_rate",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-02:S5"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-02:R5_deeptrace_audit",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-02:S6"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-02:R6_surface_quality_vs_fact_support",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-02:S7"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-02:R7_search_depth_ablation",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-02:S7"

},

{

"dimension": "model_profile",

"from": "report:V2-SOL-03",

"status": "requested-not-independent",

"to": "model-profile:gpt-5.6-sol"

},

{

"dimension": "prompt",

"from": "report:V2-SOL-03",

"status": "shared-exact-bytes",

"to": "prompt:sha256:d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670"

},

{

"dimension": "retrieval_infrastructure",

"from": "report:V2-SOL-03",

"status": "unknown-or-shared",

"to": "retrieval:codex-public-web-implementation-unknown"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-03:s1",

"status": "exact-normalized-match",

"to": "span:sha256:46c392bc942cd88d525d74298e6bc6ebd9aeaec533ddfb7b7b761a9f5b454b05"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-03:s1",

"status": "inaccessible",

"to": "https://onlinelibrary.wiley.com/doi/10.1111/jdv.70035"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-03:s1",

"status": "normalized",

"to": "work:keplinger-dermatology-audit"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-03:s1",

"status": "examined",

"to": "edition:keplinger-vor-2025"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-03:s2",

"status": "exact-normalized-match",

"to": "span:sha256:2f311bfb451c6208101cb7583af37c338914fcdb10f5c75b131e23d5bb08e66a"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-03:s2",

"status": "retrieved",

"to": "https://arxiv.org/html/2506.11763"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-03:s2",

"status": "normalized",

"to": "work:deepresearch-bench-paper"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-03:s2",

"status": "examined",

"to": "edition:drbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-03:s3",

"status": "unresolved",

"to": "span:sha256:91dd7d34a9546d71c8789056ebde06de25247281b1fb03ff41fc3a3ff20e9520"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-03:s3",

"status": "retrieved",

"to": "https://arxiv.org/html/2509.04499"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-03:s3",

"status": "normalized",

"to": "work:deeptrace"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-03:s3",

"status": "examined",

"to": "edition:deeptrace-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-03:s4",

"status": "unresolved",

"to": "span:sha256:c130950377ad7bcb4218863dbbb72f86c43005d543056438e07501f32d369df7"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-03:s4",

"status": "unresolved",

"to": "span:sha256:814248358095099e8d04475d7f17ce5aff8b4fe64f11239ea48edaf4d9a4b030"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-03:s4",

"status": "exact-normalized-match",

"to": "span:sha256:b68153666e4e12aee8dc887fef0a61fa5f44b5f680a0b7eb95ff3defcf8b5bf1"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-03:s4",

"status": "retrieved",

"to": "https://arxiv.org/html/2604.03173"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-03:s4",

"status": "normalized",

"to": "work:url-health"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-03:s4",

"status": "examined",

"to": "edition:url-health-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-03:s5",

"status": "exact-normalized-match",

"to": "span:sha256:bb16d5f06abe8641d9ea6694e549aa15e240dd88946aa47a2d4e30ddd3351a24"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-03:s5",

"status": "exact-normalized-match",

"to": "span:sha256:6e961498ed0f81a950f775956587f741e0eb9f11640fd3cb40870585558d869d"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-03:s5",

"status": "retrieved",

"to": "https://arxiv.org/html/2605.06635"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-03:s5",

"status": "normalized",

"to": "work:cited-not-verified"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-03:s5",

"status": "examined",

"to": "edition:cnv-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-03:s6",

"status": "exact-normalized-match",

"to": "span:sha256:6885cf2524462022475b5829da081b2d35fcfe7463ca6ad0979e7c4fb1f0b54d"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-03:s6",

"status": "retrieved",

"to": "https://arxiv.org/abs/2607.08700"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-03:s6",

"status": "normalized",

"to": "work:citation-verifier-benchmark"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-03:s6",

"status": "examined",

"to": "edition:citation-verifier-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-03:s7",

"status": "exact-normalized-match",

"to": "span:sha256:e5bcf0241a1aeae7d9a92755b1bcf0adeeea93a8922f22f832693c61780386e7"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-03:s7",

"status": "retrieved",

"to": "https://github.com/Ayanami0730/deep_research_bench"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-03:s7",

"status": "normalized",

"to": "work:deepresearch-bench-repository"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-03:s7",

"status": "examined",

"to": "edition:drbench-repo-main-469cce5"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-03:r1_reference_metadata_human_audit",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-03:s1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-03:r2_claim_support_human_audit",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-03:s1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-03:r3_deepresearch_bench_fact",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-03:s2"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-03:r3_deepresearch_bench_fact",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-03:s7"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-03:r4_deeptrace_support",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-03:s3"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-03:r5_deep_agent_url_resolution",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-03:s4"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-03:r5_deep_agent_url_resolution",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-03:s2"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-03:r6_url_self_correction",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-03:s4"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-03:r7_source_attribution_framework",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-03:s5"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-03:r8_search_depth_ablation",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-03:s5"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-03:r9_verifier_calibration",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-03:s6"

},

{

"dimension": "model_profile",

"from": "report:V2-SOL-04",

"status": "requested-not-independent",

"to": "model-profile:gpt-5.6-sol"

},

{

"dimension": "prompt",

"from": "report:V2-SOL-04",

"status": "shared-exact-bytes",

"to": "prompt:sha256:d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670"

},

{

"dimension": "retrieval_infrastructure",

"from": "report:V2-SOL-04",

"status": "unknown-or-shared",

"to": "retrieval:codex-public-web-implementation-unknown"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S1",

"status": "exact-normalized-match",

"to": "span:sha256:74f6acafbb4c54d556a572460bf1b45bacd1e293f4aa16cc2d499174096b9c20"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S1",

"status": "exact-normalized-match",

"to": "span:sha256:deada8bacb987a47f2a42e52fbfeb5c1beb3d250c0be79e2f9c5cdd6e16b4bc7"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S1",

"status": "exact-normalized-match",

"to": "span:sha256:417182b4c15ac550469e435a02aab1e91b395f72f186312de9d34d57fd1c6c98"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-04:S1",

"status": "retrieved",

"to": "https://arxiv.org/html/2605.06635v1"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-04:S1",

"status": "normalized",

"to": "work:cited-not-verified"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-04:S1",

"status": "examined",

"to": "edition:cnv-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S2",

"status": "exact-normalized-match",

"to": "span:sha256:196c54d6212fbc0c054dbf8e8467f2388c1f788bb93f4b1a4b6ae085eed17685"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S2",

"status": "exact-normalized-match",

"to": "span:sha256:13cb0ddb087b82988b7445a9cfb02562f9c044ef4f244e8b176ff475859eaea3"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S2",

"status": "exact-normalized-match",

"to": "span:sha256:b83996e3b5fdf878e04d6d41d0e7a1eebee9fdd9bbccf2ae4eaf33524fcfca44"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S2",

"status": "exact-normalized-match",

"to": "span:sha256:f35c9565f2fd047f6e7272dc608bbf6a63813a717bad5354c24bdaac65e40bfe"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-04:S2",

"status": "retrieved",

"to": "https://arxiv.org/html/2506.11763v1"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-04:S2",

"status": "normalized",

"to": "work:deepresearch-bench-paper"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-04:S2",

"status": "examined",

"to": "edition:drbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S3",

"status": "exact-normalized-match",

"to": "span:sha256:6c1d327e58ef001533ea1c11010e974436bbeafbde138efcc7b516281f6d25a5"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S3",

"status": "exact-normalized-match",

"to": "span:sha256:3015a109e7898cb720e426e9a57c526277b648c75f1fb4b13631b00524eac943"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-04:S3",

"status": "retrieved",

"to": "https://arxiv.org/html/2508.15804v1"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-04:S3",

"status": "normalized",

"to": "work:reportbench"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-04:S3",

"status": "examined",

"to": "edition:reportbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S4",

"status": "exact-normalized-match",

"to": "span:sha256:8cce3d9ee1e1674ba1b93321dd9b99a55d834ca980451f8ab61bcaf5bdda8e4d"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S4",

"status": "exact-normalized-match",

"to": "span:sha256:9dd6eb46529e77a62748093528e52685fbd92d46a756aef750d22aa25981db23"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-04:S4",

"status": "retrieved",

"to": "https://arxiv.org/html/2509.04499v1"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-04:S4",

"status": "normalized",

"to": "work:deeptrace"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-04:S4",

"status": "examined",

"to": "edition:deeptrace-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S5",

"status": "exact-normalized-match",

"to": "span:sha256:8c8d9a6dc003bf0298542b85df3f6f08db631e68e89f606f4e53e36e86dde167"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S5",

"status": "exact-normalized-match",

"to": "span:sha256:39a9931f05ea356fe904ed15768fa08c66440b1fcb20f98e267a315e6d562af9"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S5",

"status": "exact-normalized-match",

"to": "span:sha256:f05b3189dc3fafd45b5bde643a91fd0616832928215a37a2430a5a658e592029"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-04:S5",

"status": "retrieved",

"to": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-04:S5",

"status": "normalized",

"to": "work:keplinger-dermatology-audit"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-04:S5",

"status": "examined",

"to": "edition:keplinger-vor-2025"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S6",

"status": "exact-normalized-match",

"to": "span:sha256:94ad583514aa7e04b8d5f7b8f9cdda6fdffe8ca6712f033da3ce4f18fe7c03f1"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-04:S6",

"status": "retrieved",

"to": "https://data.mendeley.com/datasets/3s73z9zf3c/1"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-04:S6",

"status": "normalized",

"to": "work:keplinger-supplement"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-04:S6",

"status": "examined",

"to": "edition:keplinger-supplement-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S7",

"status": "exact-normalized-match",

"to": "span:sha256:e54ee1457854a9a5e46f7c7c2e66ac6a5a69b963f8ee3450f40052f138d5d0c1"

},

{

"dimension": "exact_span",

"from": "citation:V2-SOL-04:S7",

"status": "exact-normalized-match",

"to": "span:sha256:794bdf7e0595799afbb0d55f3dc522077aed9615c59f1e3575707148cf444bcf"

},

{

"dimension": "requested_url",

"from": "citation:V2-SOL-04:S7",

"status": "retrieved",

"to": "https://arxiv.org/abs/2607.08700"

},

{

"dimension": "source_work",

"from": "citation:V2-SOL-04:S7",

"status": "normalized",

"to": "work:citation-verifier-benchmark"

},

{

"dimension": "edition",

"from": "citation:V2-SOL-04:S7",

"status": "examined",

"to": "edition:citation-verifier-arxiv-v1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-04:R1",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-04:S1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-04:R2",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-04:S1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-04:R3",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-04:S2"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-04:R4",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-04:S3"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-04:R5",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-04:S4"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-04:R6",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-04:S5"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-04:R6",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-04:S6"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-04:R7",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-04:S5"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-04:R7",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-SOL-04:S6"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-SOL-04:R8",

"status": "declared-and-resolved",

"to": "citation:V2-SOL-04:S7"

},

{

"dimension": "model_profile",

"from": "report:V2-TERRA-01",

"status": "requested-not-independent",

"to": "model-profile:gpt-5.6-terra"

},

{

"dimension": "prompt",

"from": "report:V2-TERRA-01",

"status": "shared-exact-bytes",

"to": "prompt:sha256:d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670"

},

{

"dimension": "retrieval_infrastructure",

"from": "report:V2-TERRA-01",

"status": "unknown-or-shared",

"to": "retrieval:codex-public-web-implementation-unknown"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s1_deepresearchbench",

"status": "ordered-fragment-match",

"to": "span:sha256:b0a39c963cdfdc5f9852cfc9f7e33e05d31232bd1d8998851e879c38e6c34843"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s1_deepresearchbench",

"status": "unresolved",

"to": "span:sha256:77001af3bf7a976754d0b5d149da1c4e9b464c3515694958fe01a3c4ca8c7a5c"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s1_deepresearchbench",

"status": "unresolved",

"to": "span:sha256:34a61ece593714d9608073ee5aeea22635a3fb111582fc8e2e8bf0b371955b70"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-01:s1_deepresearchbench",

"status": "retrieved",

"to": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-01:s1_deepresearchbench",

"status": "normalized",

"to": "work:deepresearch-bench-paper"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-01:s1_deepresearchbench",

"status": "examined",

"to": "edition:drbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s2_reportbench",

"status": "unresolved",

"to": "span:sha256:f4c499ea114f2647a0dc4830b72e4b9120d6c368f846efb1ccd3ae4c0ed4f2c7"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s2_reportbench",

"status": "unresolved",

"to": "span:sha256:a6bde87f202e383e8bdd8403b2e5241b65380e0616e61b18f36a7c3719b55d76"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s2_reportbench",

"status": "ordered-fragment-match",

"to": "span:sha256:7a6046a756d6ec0315d330cf0817d062ea48498bdc9c90b3489d64b2bd12a3a4"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-01:s2_reportbench",

"status": "retrieved",

"to": "https://arxiv.org/pdf/2508.15804"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-01:s2_reportbench",

"status": "normalized",

"to": "work:reportbench"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-01:s2_reportbench",

"status": "examined",

"to": "edition:reportbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s3_researcherbench",

"status": "unresolved",

"to": "span:sha256:5351ba5f0c5afba376e1a70ffafab3d6a4eefae1d9507b5206a64951f360a095"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s3_researcherbench",

"status": "unresolved",

"to": "span:sha256:5fb35a076c774fd7141fd642f5490f3679fc2452e9e886b62d2e3c370b9c029d"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s3_researcherbench",

"status": "unresolved",

"to": "span:sha256:eadcf02c45f4b152c73c6f871d116e7d1db8e78c796671580a5eb72e7ff16687"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-01:s3_researcherbench",

"status": "retrieved",

"to": "https://arxiv.org/pdf/2507.16280"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-01:s3_researcherbench",

"status": "normalized",

"to": "work:researcherbench"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-01:s3_researcherbench",

"status": "examined",

"to": "edition:researcherbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s4_cited_not_verified",

"status": "unresolved",

"to": "span:sha256:48a692e799a44e37783d312cb80a0500cca0aa94abb30dc074c1cb6f0bb0ead1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s4_cited_not_verified",

"status": "unresolved",

"to": "span:sha256:a9625285d829a0463b566e89d328e49b9c21cc07e83dbf845ac422f5a8e4b72e"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-01:s4_cited_not_verified",

"status": "retrieved",

"to": "https://arxiv.org/pdf/2605.06635"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-01:s4_cited_not_verified",

"status": "normalized",

"to": "work:cited-not-verified"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-01:s4_cited_not_verified",

"status": "examined",

"to": "edition:cnv-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s5_urlhealth",

"status": "ordered-fragment-match",

"to": "span:sha256:55de6a6ca9414a252592e9f77544a9259533dd713ce1f12f6a216752a2b76e8f"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s5_urlhealth",

"status": "unresolved",

"to": "span:sha256:6af594d0155420f020802bee4a4dae0cd19f46d2c69dc107c69e75e298299169"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-01:s5_urlhealth",

"status": "unresolved",

"to": "span:sha256:baa7644423c27da0263b753522f1c14915dfdb8f9505577f8ff755a0e668c023"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-01:s5_urlhealth",

"status": "retrieved",

"to": "https://arxiv.org/pdf/2604.03173"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-01:s5_urlhealth",

"status": "normalized",

"to": "work:url-health"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-01:s5_urlhealth",

"status": "examined",

"to": "edition:url-health-arxiv-v1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-01:r1_deepresearchbench_fact",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-01:s1_deepresearchbench"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-01:r2_reportbench_cited_statement_match",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-01:s2_reportbench"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-01:r3_researcherbench_faithfulness_and_groundedness",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-01:s3_researcherbench"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-01:r4_cited_not_verified_source_attribution",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-01:s4_cited_not_verified"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-01:r5_urlhealth_resolution",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-01:s5_urlhealth"

},

{

"dimension": "model_profile",

"from": "report:V2-TERRA-02",

"status": "requested-not-independent",

"to": "model-profile:gpt-5.6-terra"

},

{

"dimension": "prompt",

"from": "report:V2-TERRA-02",

"status": "shared-exact-bytes",

"to": "prompt:sha256:d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670"

},

{

"dimension": "retrieval_infrastructure",

"from": "report:V2-TERRA-02",

"status": "unknown-or-shared",

"to": "retrieval:codex-public-web-implementation-unknown"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s1_deepresearchbench",

"status": "exact-normalized-match",

"to": "span:sha256:eaf37bef7b8dc35f3e00b85d3e58ad872dc0ea604a062c8cdd8929184587d5d9"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s1_deepresearchbench",

"status": "unresolved",

"to": "span:sha256:b74af48901531355cd4351211eac41aef691ea4eb29488347782fca1ec2d1eb8"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s1_deepresearchbench",

"status": "exact-normalized-match",

"to": "span:sha256:e4a9f30effc031388ed5c9993dc6f071903bec5e3f1998db4c08ab00b284b27f"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s1_deepresearchbench",

"status": "exact-normalized-match",

"to": "span:sha256:f98f1274859ba98cc538e3be6d11d345a5382cfcb9516aa1cf3f3278ea646b92"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-02:s1_deepresearchbench",

"status": "retrieved",

"to": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-02:s1_deepresearchbench",

"status": "normalized",

"to": "work:deepresearch-bench-paper"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-02:s1_deepresearchbench",

"status": "examined",

"to": "edition:drbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s2_liveresearchbench",

"status": "ordered-fragment-match",

"to": "span:sha256:d880991ad784390029d49f84949d083c4789d67d45893d87cfb6b5a38703453f"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s2_liveresearchbench",

"status": "unresolved",

"to": "span:sha256:256817dca0b5aac5741ec51862ec8d0f9b4b93ea78186a4041052c7b2d1c4382"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s2_liveresearchbench",

"status": "exact-normalized-match",

"to": "span:sha256:e6f447129d9a662fb202d4ac22e16822bacfdbcbcdb7afe3b589e21802890b8a"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-02:s2_liveresearchbench",

"status": "retrieved",

"to": "https://arxiv.org/pdf/2510.14240v5"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-02:s2_liveresearchbench",

"status": "normalized",

"to": "work:liveresearchbench"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-02:s2_liveresearchbench",

"status": "examined",

"to": "edition:liveresearchbench-arxiv-v5"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s3_deeptrace",

"status": "unresolved",

"to": "span:sha256:4d92f4721bb3d81403154a23d7fde5b342251d50a2d1034ec6d4542a35e9fc8b"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s3_deeptrace",

"status": "unresolved",

"to": "span:sha256:7572651679e7e8a0a1a6e394504e75749ae2649cfb55f7a80531c8b53acb31dc"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s3_deeptrace",

"status": "exact-normalized-match",

"to": "span:sha256:9cb1700c08368feb234207045354cbf559d6ed80f2b0bae814d2b5b3c858b73c"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s3_deeptrace",

"status": "exact-normalized-match",

"to": "span:sha256:aa1534ce2e177153ab85e512deb8d064d14ec1eae16d8032bdb5043252738d4c"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-02:s3_deeptrace",

"status": "retrieved",

"to": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-02:s3_deeptrace",

"status": "normalized",

"to": "work:deeptrace"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-02:s3_deeptrace",

"status": "examined",

"to": "edition:deeptrace-iclr-2026"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s4_researcherbench",

"status": "exact-normalized-match",

"to": "span:sha256:886b50d94e3facabc487342cd2af5157cf956622ecfc368684596c5f6b61c350"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-02:s4_researcherbench",

"status": "unresolved",

"to": "span:sha256:f85edb1051cbc19ee38bc796ed3392abf5e44fc869d59d7e82bfe2cb81696e9b"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-02:s4_researcherbench",

"status": "retrieved",

"to": "https://arxiv.org/html/2507.16280"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-02:s4_researcherbench",

"status": "normalized",

"to": "work:researcherbench"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-02:s4_researcherbench",

"status": "examined",

"to": "edition:researcherbench-arxiv-v1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-02:r1_deepresearchbench_fact",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-02:s1_deepresearchbench"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-02:r2_liveresearchbench_wide_info",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-02:s2_liveresearchbench"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-02:r3_liveresearchbench_market_analysis",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-02:s2_liveresearchbench"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-02:r4_deeptrace",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-02:s3_deeptrace"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-02:r5_researcherbench",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-02:s4_researcherbench"

},

{

"dimension": "model_profile",

"from": "report:V2-TERRA-03",

"status": "requested-not-independent",

"to": "model-profile:gpt-5.6-terra"

},

{

"dimension": "prompt",

"from": "report:V2-TERRA-03",

"status": "shared-exact-bytes",

"to": "prompt:sha256:d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670"

},

{

"dimension": "retrieval_infrastructure",

"from": "report:V2-TERRA-03",

"status": "unknown-or-shared",

"to": "retrieval:codex-public-web-implementation-unknown"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S1",

"status": "exact-normalized-match",

"to": "span:sha256:cd21c450c0d33df20a2b1a539d5725347db1b5bab7d1e9866c24790d845e5873"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S1",

"status": "unresolved",

"to": "span:sha256:35357cd8e4ff1531a84b4a5c8923db97c7b44fecd19d7666f326bbf7a1d5ec45"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S1",

"status": "unresolved",

"to": "span:sha256:d004f718815ff769195d087f1159e50cf5d00261b8ef85229625a65740d55b3d"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S1",

"status": "unresolved",

"to": "span:sha256:ec8d8cf5ad9b690ecd983e4d5c673d9624200351579e72868291abc15d0f2949"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S1",

"status": "unresolved",

"to": "span:sha256:fe6b18fa48b171b5bd7c74fda45f38ed1b087ee5035f6d35dcf8c701b3033ea8"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S1",

"status": "ordered-fragment-match",

"to": "span:sha256:5cd9ba4ca72da10cf51c0beddc6ea2765c74721bdd526a67156f44754219f440"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-03:S1",

"status": "retrieved",

"to": "https://arxiv.org/html/2605.06635"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-03:S1",

"status": "normalized",

"to": "work:cited-not-verified"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-03:S1",

"status": "examined",

"to": "edition:cnv-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S2",

"status": "exact-normalized-match",

"to": "span:sha256:015bb68c904bf225e84c82b1acffb3cc38a4a087d7b0f79d3b0102b685b2265e"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S2",

"status": "ordered-fragment-match",

"to": "span:sha256:0d4f357f2a5565bdaea01a1e537650958d71216868d1a2b0bc14c47a45b6d0ef"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S2",

"status": "unresolved",

"to": "span:sha256:3ec266d89a7c774ebcee3d2b8b758ac56176c382481cd0d2f0bd893aaa142281"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S2",

"status": "unresolved",

"to": "span:sha256:de53fab3c5b2e262ba56cdd5a27eb002c2b4641755ccc38c734a5bd506a670ef"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-03:S2",

"status": "retrieved",

"to": "https://arxiv.org/html/2506.11763"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-03:S2",

"status": "normalized",

"to": "work:deepresearch-bench-paper"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-03:S2",

"status": "examined",

"to": "edition:drbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S3",

"status": "unresolved",

"to": "span:sha256:42a4184b2e116330dc263a3d8e9b77f4b88c9da1c325fc67c4375df0f818a514"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S3",

"status": "exact-normalized-match",

"to": "span:sha256:ba6b8a8b7d9121ebb05d594176ceed3d1e9dbc637876783abfd0141ebbd0a822"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S3",

"status": "unresolved",

"to": "span:sha256:7eefee71f11e6f5ec8c2e70272fd9189464650e68b904d9fa9c6d059bedc4195"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S3",

"status": "exact-normalized-match",

"to": "span:sha256:1db12b68cfeb4c5f5a96fa9cd2e2bd46e8dfb4f096f73154481d29e22a4d5414"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-03:S3",

"status": "retrieved",

"to": "https://arxiv.org/html/2508.15804"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-03:S3",

"status": "normalized",

"to": "work:reportbench"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-03:S3",

"status": "examined",

"to": "edition:reportbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S4",

"status": "exact-normalized-match",

"to": "span:sha256:b38a2af5982b4c85bc205d2f533a23ed3a8f40a49b651a5f00ec6932dcc67f7d"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S4",

"status": "exact-normalized-match",

"to": "span:sha256:b6a6ba48a2c4e352c8565b671a15a90400739f2aeefda8a65a31ad8a1ff9a4e4"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S4",

"status": "unresolved",

"to": "span:sha256:f8ff0887370fedb6d6676d1eaf59f86461fe5083d20dfa5b28f8ba94175ed031"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-03:S4",

"status": "retrieved",

"to": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-03:S4",

"status": "normalized",

"to": "work:keplinger-dermatology-audit"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-03:S4",

"status": "examined",

"to": "edition:keplinger-vor-2025"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S5",

"status": "ordered-fragment-match",

"to": "span:sha256:ef637b38f36195bea09205d532e8744daf3e257f93d97144f061024092bf67fe"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S5",

"status": "exact-normalized-match",

"to": "span:sha256:92eb7cf19745e24190dda84c77d590941134d09f06805c0fb605207fd942ae11"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-03:S5",

"status": "retrieved",

"to": "https://github.com/Ayanami0730/deep_research_bench"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-03:S5",

"status": "normalized",

"to": "work:deepresearch-bench-repository"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-03:S5",

"status": "examined",

"to": "edition:drbench-repo-main-469cce5"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-03:S6",

"status": "unresolved",

"to": "span:sha256:fdc35d731d751e7e0691158d8fed4a2934851897f9fef016b33376cde973827a"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-03:S6",

"status": "inaccessible",

"to": "https://pubmed.ncbi.nlm.nih.gov/40904191/"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-03:S6",

"status": "normalized",

"to": "work:keplinger-dermatology-audit"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-03:S6",

"status": "examined",

"to": "edition:keplinger-vor-2025"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-03:R1",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-03:S1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-03:R2",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-03:S1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-03:R3",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-03:S2"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-03:R3",

"status": "declared-and-resolved",

"to": "citation:V2-TERRA-03:S5"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-03:R4",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-03:S3"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-03:R5",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-03:S4"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-03:R5",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-03:S6"

},

{

"dimension": "model_profile",

"from": "report:V2-TERRA-04",

"status": "requested-not-independent",

"to": "model-profile:gpt-5.6-terra"

},

{

"dimension": "prompt",

"from": "report:V2-TERRA-04",

"status": "shared-exact-bytes",

"to": "prompt:sha256:d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670"

},

{

"dimension": "retrieval_infrastructure",

"from": "report:V2-TERRA-04",

"status": "unknown-or-shared",

"to": "retrieval:codex-public-web-implementation-unknown"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S1_deepresearchbench",

"status": "exact-normalized-match",

"to": "span:sha256:9fe42cb48702d8d96ebbbc0309692214ece6ec495c447c06e2be62f32ff00894"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S1_deepresearchbench",

"status": "unresolved",

"to": "span:sha256:539d8123b83aa1250f16cb7a8a4a66ece5d42099c0ae7c7d1fd877f1fbafb3bd"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S1_deepresearchbench",

"status": "unresolved",

"to": "span:sha256:a13bd114b7a394950897d477dccb999910794b70914bbc3a45fa6c4667181992"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S1_deepresearchbench",

"status": "exact-normalized-match",

"to": "span:sha256:76e9bd59b31c7ebd2ed7a2186e59753a79d3c730e0d7b4fbf0146fb39f4e9c5f"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-04:S1_deepresearchbench",

"status": "retrieved",

"to": "https://arxiv.org/html/2506.11763"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-04:S1_deepresearchbench",

"status": "normalized",

"to": "work:deepresearch-bench-paper"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-04:S1_deepresearchbench",

"status": "examined",

"to": "edition:drbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S2_researcherbench",

"status": "exact-normalized-match",

"to": "span:sha256:f96ad5daa81153a3689f52f5777fedfd70e8a87d44e7a1c9e2c22b5789a3707d"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S2_researcherbench",

"status": "unresolved",

"to": "span:sha256:5dbd4e3e48480f5b3c54ce606082c35bfd55bf40cd0be847cc03a8fd4d9993f6"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S2_researcherbench",

"status": "unresolved",

"to": "span:sha256:0d70187dbf9e3044ae9efd387faf34b29b9ab556aae6b0ea4c06e05964bfa2c7"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S2_researcherbench",

"status": "exact-normalized-match",

"to": "span:sha256:ea9cb2e1b0078c0857b789d61d0a0d26801cf6689186470588a87b98b4639796"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-04:S2_researcherbench",

"status": "retrieved",

"to": "https://arxiv.org/html/2507.16280"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-04:S2_researcherbench",

"status": "normalized",

"to": "work:researcherbench"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-04:S2_researcherbench",

"status": "examined",

"to": "edition:researcherbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S3_reportbench",

"status": "exact-normalized-match",

"to": "span:sha256:9c7a4c186cc85f185aa293a7c5a46c08f9dbbc1a9e63c449dc737447e42f8ee4"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S3_reportbench",

"status": "unresolved",

"to": "span:sha256:9a2cf22761f2ebb97eb31dfa02a32775248c3a7396b1e410565c313d30585cf3"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S3_reportbench",

"status": "exact-normalized-match",

"to": "span:sha256:ddf4320a6d302c13d28b6225eb7e79ac15583349e1b449d4bb967eb89a257a3d"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S3_reportbench",

"status": "exact-normalized-match",

"to": "span:sha256:084f11e57cf779850ea87aaf30ba1277438ff9efcb533fd2ebb09dba3665be8d"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-04:S3_reportbench",

"status": "retrieved",

"to": "https://arxiv.org/html/2508.15804"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-04:S3_reportbench",

"status": "normalized",

"to": "work:reportbench"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-04:S3_reportbench",

"status": "examined",

"to": "edition:reportbench-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S4_urlhealth",

"status": "unresolved",

"to": "span:sha256:131799471acf2f83cc3a17322a1c5cda6b5b46020bd894be54a1708db62eb14b"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S4_urlhealth",

"status": "unresolved",

"to": "span:sha256:dc108a5e00fb6ce5306fcde8fd0c36c85352dac361662fab3ad1eea89d66c048"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S4_urlhealth",

"status": "exact-normalized-match",

"to": "span:sha256:bb62928b5e0cb4e373b2f6bfeb579d8564d05da5588f8b150a89ea4b9b2ed0db"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S4_urlhealth",

"status": "exact-normalized-match",

"to": "span:sha256:59d0da64f738e1d8745f2bff3663eb89e273f93fe501afe37e2494a8e0ff31a1"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-04:S4_urlhealth",

"status": "retrieved",

"to": "https://arxiv.org/html/2604.03173"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-04:S4_urlhealth",

"status": "normalized",

"to": "work:url-health"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-04:S4_urlhealth",

"status": "examined",

"to": "edition:url-health-arxiv-v1"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S5_cited_not_verified",

"status": "exact-normalized-match",

"to": "span:sha256:732c03e69c6e142a492275cee5d1e2d474c75422c9b84aaaf273100e937338ed"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S5_cited_not_verified",

"status": "unresolved",

"to": "span:sha256:cebc739ec2bf421e0853467bfeccb3f42bda1cf05472076a1c1bc6d9fb1edfb9"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S5_cited_not_verified",

"status": "unresolved",

"to": "span:sha256:caaa3e2fbb4cef64558f2c8cc30388281e1505c9353fb775f960a5f18c816111"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S5_cited_not_verified",

"status": "exact-normalized-match",

"to": "span:sha256:d638c8b21f0e4a0f840880b6ffe0217ce1623758cc09fdccc32cacf2db3dd958"

},

{

"dimension": "exact_span",

"from": "citation:V2-TERRA-04:S5_cited_not_verified",

"status": "ordered-fragment-match",

"to": "span:sha256:c7dbd1d9e09980703ecbee8161594e4b426c1af63117b28662a7e98686107621"

},

{

"dimension": "requested_url",

"from": "citation:V2-TERRA-04:S5_cited_not_verified",

"status": "retrieved",

"to": "https://arxiv.org/html/2605.06635"

},

{

"dimension": "source_work",

"from": "citation:V2-TERRA-04:S5_cited_not_verified",

"status": "normalized",

"to": "work:cited-not-verified"

},

{

"dimension": "edition",

"from": "citation:V2-TERRA-04:S5_cited_not_verified",

"status": "examined",

"to": "edition:cnv-arxiv-v1"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-04:R1_deepresearchbench_support",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-04:S1_deepresearchbench"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-04:R2_researcherbench_support",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-04:S2_researcherbench"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-04:R3_reportbench_support",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-04:S3_reportbench"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-04:R4_urlhealth_resolution",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-04:S4_urlhealth"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-04:R5_cited_not_verified_support_and_resolution",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-04:S5_cited_not_verified"

},

{

"dimension": "upstream_citation",

"from": "claim:V2-TERRA-04:R6_search_depth_ablation",

"status": "declared-but-citation-unresolved",

"to": "citation:V2-TERRA-04:S5_cited_not_verified"

}

],

"editions": [

{

"canonical_url": "https://arxiv.org/html/2607.08700v1",

"edition_id": "edition:citation-verifier-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"source_text_file": "citation-verifier.txt",

"work_id": "work:citation-verifier-benchmark"

},

{

"canonical_url": "https://arxiv.org/html/2605.06635v1",

"edition_id": "edition:cnv-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"source_text_file": "cited-not-verified.txt",

"work_id": "work:cited-not-verified"

},

{

"canonical_url": "https://arxiv.org/html/2509.04499v1",

"edition_id": "edition:deeptrace-arxiv-v1",

"license": "arXiv non-exclusive distribution license",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"source_text_file": "deeptrace-v1.txt",

"work_id": "work:deeptrace"

},

{

"canonical_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"edition_id": "edition:deeptrace-iclr-2026",

"license": "unknown",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"source_text_file": "deeptrace-iclr.txt",

"work_id": "work:deeptrace"

},

{

"canonical_url": "https://arxiv.org/html/2506.11763v1",

"edition_id": "edition:drbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"source_text_file": "drbench-v1.txt",

"work_id": "work:deepresearch-bench-paper"

},

{

"canonical_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/465f22be10e07b301c6ed58f0472f704-Paper-Conference.pdf",

"edition_id": "edition:drbench-iclr-2026",

"license": "unknown",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"source_text_file": "drbench-iclr.txt",

"work_id": "work:deepresearch-bench-paper"

},

{

"canonical_url": "https://github.com/Ayanami0730/deep_research_bench/commit/469cce54ea7f6a63c163d3d9fec879cf289ec484",

"edition_id": "edition:drbench-repo-main-469cce5",

"license": "Apache-2.0",

"license_treatment": "metadata and quote-minimal README spans",

"source_text_file": "drbench-repo.txt",

"work_id": "work:deepresearch-bench-repository"

},

{

"canonical_url": "https://data.mendeley.com/datasets/3s73z9zf3c/1",

"edition_id": "edition:keplinger-supplement-v1",

"license": "CC BY 4.0",

"license_treatment": "metadata and quote-minimal landing-page spans",

"source_text_file": "mendeley-v1.txt",

"work_id": "work:keplinger-supplement"

},

{

"canonical_url": "https://data.mendeley.com/datasets/3s73z9zf3c/2",

"edition_id": "edition:keplinger-supplement-v2",

"license": "CC BY 4.0",

"license_treatment": "metadata and quote-minimal landing-page spans; file API required authentication and was not used",

"source_text_file": "mendeley-v2-page.txt",

"work_id": "work:keplinger-supplement"

},

{

"canonical_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"edition_id": "edition:keplinger-vor-2025",

"license": "CC BY-NC 4.0",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"source_text_file": "keplinger.txt",

"work_id": "work:keplinger-dermatology-audit"

},

{

"canonical_url": "https://arxiv.org/pdf/2510.14240v5",

"edition_id": "edition:liveresearchbench-arxiv-v5",

"license": "CC BY-NC-SA 4.0",

"license_treatment": "quote-minimal attributed spans; raw CC BY claim is corrected in review records",

"source_text_file": "liveresearchbench.txt",

"work_id": "work:liveresearchbench"

},

{

"canonical_url": "https://arxiv.org/html/2508.15804v1",

"edition_id": "edition:reportbench-arxiv-v1",

"license": "CC BY 4.0",

"license_treatment": "quote-minimal attributed spans",

"source_text_file": "reportbench.txt",

"work_id": "work:reportbench"

},

{

"canonical_url": "https://arxiv.org/html/2507.16280v1",

"edition_id": "edition:researcherbench-arxiv-v1",

"license": "arXiv non-exclusive distribution license",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"source_text_file": "researcherbench.txt",

"work_id": "work:researcherbench"

},

{

"canonical_url": "https://arxiv.org/html/2604.03173v1",

"edition_id": "edition:url-health-arxiv-v1",

"license": "CC0 1.0",

"license_treatment": "quote-minimal attributed spans",

"source_text_file": "urlhealth.txt",

"work_id": "work:url-health"

}

],

"evidence_cutoff": "2026-08-22",

"limitations": [

"The eight reports share exact prompt bytes, Codex runtime lineage, and unknown or shared retrieval infrastructure; they are observations, not independent evidence roots.",

"Candidate warrant roots are author-side normalization results and remain zero independently confirmed roots until a fresh-clone reviewer repeats source and count checks.",

"HTTP readback, text matching, and semantic warrant are separate gates.",

"No universal claim about all agents, current products, or all web sources is authorized."

],

"protocol_id": "em-0026-agent-citation-trace-v2",

"readback_captured_at": "2026-08-23T16:28:41Z",

"reports": [

{

"answer_bytes": 40476,

"answer_path": "research/how-we-know/agent-citation-lineage/answers-v2/V2-SOL-01.json",

"answer_sha256": "17a7cf97b717d7b520026b9eb52a7da59ddb9b5f698c2aeadfab3f38c91008a0",

"prompt_sha256": "d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670",

"reported_model_identity": "unknown",

"requested_model_profile": "gpt-5.6-sol",

"retrieval_infrastructure": "unknown",

"run_id": "V2-SOL-01",

"status": "completed",

"trace_path": "research/how-we-know/agent-citation-lineage/traces-v2/V2-SOL-01.json"

},

{

"answer_bytes": 32376,

"answer_path": "research/how-we-know/agent-citation-lineage/answers-v2/V2-SOL-02.json",

"answer_sha256": "f72975b2b98532116ca9d2c510e2343df505bfce359ebfef595a0a35b77cb810",

"prompt_sha256": "d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670",

"reported_model_identity": "unknown",

"requested_model_profile": "gpt-5.6-sol",

"retrieval_infrastructure": "unknown",

"run_id": "V2-SOL-02",

"status": "completed",

"trace_path": "research/how-we-know/agent-citation-lineage/traces-v2/V2-SOL-02.json"

},

{

"answer_bytes": 38504,

"answer_path": "research/how-we-know/agent-citation-lineage/answers-v2/V2-SOL-03.json",

"answer_sha256": "782a60cc19d88217694cd672bac6f659f44fa0bd0f006f64a65ea77386d82dc7",

"prompt_sha256": "d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670",

"reported_model_identity": "unknown",

"requested_model_profile": "gpt-5.6-sol",

"retrieval_infrastructure": "unknown",

"run_id": "V2-SOL-03",

"status": "completed",

"trace_path": "research/how-we-know/agent-citation-lineage/traces-v2/V2-SOL-03.json"

},

{

"answer_bytes": 36500,

"answer_path": "research/how-we-know/agent-citation-lineage/answers-v2/V2-SOL-04.json",

"answer_sha256": "216552854069e32a8dc4986c57f59dc3b9b56d44f01d527167b3f311384a9261",

"prompt_sha256": "d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670",

"reported_model_identity": "unknown",

"requested_model_profile": "gpt-5.6-sol",

"retrieval_infrastructure": "credential-free public web search, open, click, and find operations; no logged-in browser session or account connector",

"run_id": "V2-SOL-04",

"status": "completed",

"trace_path": "research/how-we-know/agent-citation-lineage/traces-v2/V2-SOL-04.json"

},

{

"answer_bytes": 22598,

"answer_path": "research/how-we-know/agent-citation-lineage/answers-v2/V2-TERRA-01.json",

"answer_sha256": "e90c36b6f172fbe291d75ce1ea9451b1160515a2b7b4e0ead66aa1a76cc8b735",

"prompt_sha256": "d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670",

"reported_model_identity": "unknown",

"requested_model_profile": "gpt-5.6-terra",

"retrieval_infrastructure": "unknown",

"run_id": "V2-TERRA-01",

"status": "completed",

"trace_path": "research/how-we-know/agent-citation-lineage/traces-v2/V2-TERRA-01.json"

},

{

"answer_bytes": 20907,

"answer_path": "research/how-we-know/agent-citation-lineage/answers-v2/V2-TERRA-02.json",

"answer_sha256": "819d320162a09909a73df00a4e1ebbdd42e67794cb02e3f7c379f601e10a0192",

"prompt_sha256": "d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670",

"reported_model_identity": "unknown",

"requested_model_profile": "gpt-5.6-terra",

"retrieval_infrastructure": "unknown",

"run_id": "V2-TERRA-02",

"status": "completed",

"trace_path": "research/how-we-know/agent-citation-lineage/traces-v2/V2-TERRA-02.json"

},

{

"answer_bytes": 23183,

"answer_path": "research/how-we-know/agent-citation-lineage/answers-v2/V2-TERRA-03.json",

"answer_sha256": "e211dfe49066bb9f4ff9e59b91173d3b0f45b7b1e8ca5135c7435f3882d7e264",

"prompt_sha256": "d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670",

"reported_model_identity": "unknown",

"requested_model_profile": "gpt-5.6-terra",

"retrieval_infrastructure": "unknown",

"run_id": "V2-TERRA-03",

"status": "completed",

"trace_path": "research/how-we-know/agent-citation-lineage/traces-v2/V2-TERRA-03.json"

},

{

"answer_bytes": 30189,

"answer_path": "research/how-we-know/agent-citation-lineage/answers-v2/V2-TERRA-04.json",

"answer_sha256": "8deffa91b6bb52d1de4ef4965d0087eca3d3a8a96d971a5e89753c1928cc4ef4",

"prompt_sha256": "d321a9cec7b5fe419157c0623e18ff0020cb0080fbe6dd9ed4fd20a0b896f670",

"reported_model_identity": "unknown",

"requested_model_profile": "gpt-5.6-terra",

"retrieval_infrastructure": "unknown",

"run_id": "V2-TERRA-04",

"status": "completed",

"trace_path": "research/how-we-know/agent-citation-lineage/traces-v2/V2-TERRA-04.json"

}

],

"schema": "https://epistemedia.org/research/agent-citation-evidence-ledger-v1.json",

"spans": [

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "body, paragraph beginning 'Taking ChatGPT as an example'; Table 1",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "23 (100.0%) ... All correct ... 16 (69.6%)",

"quote_sha256": "94db53a0356c79519c78f1f5bd01799894b83babb1fc3ced35f740a68a5b3b79",

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"run_id": "V2-SOL-01",

"source_id": "s1_keplinger",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "sp1a",

"span_occurrence_id": "V2-SOL-01:s1_keplinger:sp1a",

"span_root_id": "span:sha256:c29518951f6335c3c613ebed9c4babc2bc1903426992cd90118a999be69b5cdd",

"supports_declared_by_raw_answer": "ChatGPT Deep Research reference-list denominator and entirely correct metadata numerator/rate"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "body, paragraph immediately after Table 1",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "high error rates: 51.3 \u00b1 6.5% for ChatGPT and 57.8 \u00b1 22.7% for Le Chat",

"quote_sha256": "2e43a0652065a6a1fb69de224082ac79d903cc71cac0664a1cb376cbfa09d0e8",

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"run_id": "V2-SOL-01",

"source_id": "s1_keplinger",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "sp1b",

"span_occurrence_id": "V2-SOL-01:s1_keplinger:sp1b",

"span_root_id": "span:sha256:b903dc2284e1b4d80bb6ef948d79bbdbc7d7447e871281b8577284883f75ccfa",

"supports_declared_by_raw_answer": "human-reviewed citation-bearing-sentence inaccuracy rates"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "Figure 1 caption",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "Sentences may bear multiple types of inaccuracy, simultaneously.",

"quote_sha256": "05af5f5f154c4bed8267aac71674219aad878b9ec2635ab85e3ae60474cd02c5",

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"run_id": "V2-SOL-01",

"source_id": "s1_keplinger",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "sp1c",

"span_occurrence_id": "V2-SOL-01:s1_keplinger:sp1c",

"span_root_id": "span:sha256:0bea3f00dfce1763dae86c14901bee21053f2d8ceee7f9dde4ab33ec78d3eb50",

"supports_declared_by_raw_answer": "the reported categories overlap and are not additive"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-supplement-v2",

"license_treatment": "metadata and quote-minimal landing-page spans; file API required authentication and was not used",

"locator": "dataset page, Description",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "Prompts for Deep Research-generated reviews and evaluation results",

"quote_sha256": "c5ab0cef07fb26a5758cd8b25a249016665c446b524e1443a10c4c8929938d21",

"requested_url": "https://data.mendeley.com/datasets/3s73z9zf3c/2",

"run_id": "V2-SOL-01",

"source_id": "s1a_keplinger_data",

"source_text_sha256": "2e96cc4ae8e28c60fcb94dd40ef64a15252e4416e7bf33185b8b76ea6ff228e4",

"source_work_id": "work:keplinger-supplement",

"span_id": "sp1d",

"span_occurrence_id": "V2-SOL-01:s1a_keplinger_data:sp1d",

"span_root_id": "span:sha256:6f79c3c824ccf889d73e234335661e5fca68d90f8c78ee97aef488e4fdb32a49",

"supports_declared_by_raw_answer": "the official artifact includes prompts and evaluation details"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "page 4, Section 3.2, Support Judgment",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "This yields a binary judgment: 'support' or 'not support'.",

"quote_sha256": "78f021c07980218ead835a986c84780444479e627dcfa9f2ea126b3604b89ad7",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/465f22be10e07b301c6ed58f0472f704-Paper-Conference.pdf",

"run_id": "V2-SOL-01",

"source_id": "s2_drbench",

"source_text_sha256": "ced9e7b0ae4ab2c26add3230a7b40200673ee4fc0c5c763df226bfe784d60016",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "sp2a",

"span_occurrence_id": "V2-SOL-01:s2_drbench:sp2a",

"span_root_id": "span:sha256:7d7331942f9d0520e18e96206e7718e5187788be4d48cef995fac79b235db757",

"supports_declared_by_raw_answer": "FACT's semantic support decision"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "page 6, Table 1, proprietary deep-research-agent rows",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "Grok ... 73.08 ... Perplexity ... 82.63 ... Gemini ... 78.30 ... OpenAI ... 75.01",

"quote_sha256": "fa0ab638c02ed6458f5f0f70cb866f3b5cb4f09dacdd05782d6fa43d58e3a283",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/465f22be10e07b301c6ed58f0472f704-Paper-Conference.pdf",

"run_id": "V2-SOL-01",

"source_id": "s2_drbench",

"source_text_sha256": "ced9e7b0ae4ab2c26add3230a7b40200673ee4fc0c5c763df226bfe784d60016",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "sp2b",

"span_occurrence_id": "V2-SOL-01:s2_drbench:sp2b",

"span_root_id": "span:sha256:6070d8d9e7ba0d923aca7ae3517013c0e865119cea05c3e9cb0565d273eee0ee",

"supports_declared_by_raw_answer": "conference-edition citation-accuracy values"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "page 5, Section 4.1",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "complete set of 100 tasks",

"quote_sha256": "11034cf2238003f213bf87210c9017f57cc743910fa70754f3e61f875abe75e6",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/465f22be10e07b301c6ed58f0472f704-Paper-Conference.pdf",

"run_id": "V2-SOL-01",

"source_id": "s2_drbench",

"source_text_sha256": "ced9e7b0ae4ab2c26add3230a7b40200673ee4fc0c5c763df226bfe784d60016",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "sp2c",

"span_occurrence_id": "V2-SOL-01:s2_drbench:sp2c",

"span_root_id": "span:sha256:c27aaca02bda06ed764be53351158fc862af9d4a556d3e82074507436f48c3f8",

"supports_declared_by_raw_answer": "main-results task denominator"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "page 16, FACT judge validation",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "aligned with human 'support' determinations in 96% of cases",

"quote_sha256": "8be6257e74c3696813fb0079c27e93e8bec16c71d214b6da43811fbd9b406ba4",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/465f22be10e07b301c6ed58f0472f704-Paper-Conference.pdf",

"run_id": "V2-SOL-01",

"source_id": "s2_drbench",

"source_text_sha256": "ced9e7b0ae4ab2c26add3230a7b40200673ee4fc0c5c763df226bfe784d60016",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "sp2d",

"span_occurrence_id": "V2-SOL-01:s2_drbench:sp2d",

"span_root_id": "span:sha256:279db725380c133044944dccb379b75a396242219aa30437b26ddc40b09a3b5c",

"supports_declared_by_raw_answer": "human calibration of the automated support judge"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-repo-main-469cce5",

"license_treatment": "metadata and quote-minimal README spans",

"locator": "repository README, Overview",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "100 PhD-level research tasks",

"quote_sha256": "2e8262448070ca7bcc1ea305d2d87a722f34721737d455d55f62185010bd2e32",

"requested_url": "https://github.com/Ayanami0730/deep_research_bench",

"run_id": "V2-SOL-01",

"source_id": "s2a_drbench_repo",

"source_text_sha256": "afebe200cb45042b2783d6ad581770bbd8674027d4e7443e2eb6c26720c6cb0c",

"source_work_id": "work:deepresearch-bench-repository",

"span_id": "sp2e",

"span_occurrence_id": "V2-SOL-01:s2a_drbench_repo:sp2e",

"span_root_id": "span:sha256:0abbab3d6269d58eb76d8e90eff1e4c27512cf46d9b8e5ff0a9f537b18f232fa",

"supports_declared_by_raw_answer": "official project artifact and benchmark scale"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "page 6, Section 3.1.4",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "fraction of statement citations that accurately reflect that a source's content supports the statement",

"quote_sha256": "1320d1a169d78cdf9f0b87b4263fbf3ee04067daefddfa8a93722cf51ac69485",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"run_id": "V2-SOL-01",

"source_id": "s3_deeptrace",

"source_text_sha256": "0eaa415eca25bed5e591b8250640d695b86f6dac6601f5f00167ee3ac7ab9171",

"source_work_id": "work:deeptrace",

"span_id": "sp3a",

"span_occurrence_id": "V2-SOL-01:s3_deeptrace:sp3a",

"span_root_id": "span:sha256:b45b782077327a7123cff72789773ad2f70f5608c8e0e33b0d211b3d60e0de0a",

"supports_declared_by_raw_answer": "definition of citation accuracy"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "page 8, Table 1",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "GPT-5(DR) ... 79.1 ... PPLX(DR) ... 58.0 ... Gemini(DR) ... 50.3",

"quote_sha256": "b2708c9e7c59cc34ab973c79d61bad55c7cb3a0ad55a47d9e35f17d9e7c98cc3",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"run_id": "V2-SOL-01",

"source_id": "s3_deeptrace",

"source_text_sha256": "0eaa415eca25bed5e591b8250640d695b86f6dac6601f5f00167ee3ac7ab9171",

"source_work_id": "work:deeptrace",

"span_id": "sp3b",

"span_occurrence_id": "V2-SOL-01:s3_deeptrace:sp3b",

"span_root_id": "span:sha256:4e55438723cddd7dfbd56c115e5fe6e809c0091ec7631928d67490aafc49f118",

"supports_declared_by_raw_answer": "selected exact table citation-accuracy values"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "page 4, source scraping",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "For roughly 15% of the URLs, the Reader tool returns an error",

"quote_sha256": "a38c9ac275570f2b78dd1e1bf7663579e35f9037131e12d721dfb4c6f005683d",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"run_id": "V2-SOL-01",

"source_id": "s3_deeptrace",

"source_text_sha256": "0eaa415eca25bed5e591b8250640d695b86f6dac6601f5f00167ee3ac7ab9171",

"source_work_id": "work:deeptrace",

"span_id": "sp3c",

"span_occurrence_id": "V2-SOL-01:s3_deeptrace:sp3c",

"span_root_id": "span:sha256:a15ab02e9fcb23df02fb03d28967d4220ddf65ad380d223049f22b6ab68ed5eb",

"supports_declared_by_raw_answer": "source-access exclusion limitation"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "page 14, Table 3",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "Factual support (statement-source) 0.62 binary",

"quote_sha256": "bb4f28cd0f4527c633bcfacf51b18ced7387c4a2f266c59d41de7ee899b24b85",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"run_id": "V2-SOL-01",

"source_id": "s3_deeptrace",

"source_text_sha256": "0eaa415eca25bed5e591b8250640d695b86f6dac6601f5f00167ee3ac7ab9171",

"source_work_id": "work:deeptrace",

"span_id": "sp3d",

"span_occurrence_id": "V2-SOL-01:s3_deeptrace:sp3d",

"span_root_id": "span:sha256:35927851edd2e7e5e0f60d98498940f88304ba99bf1f85a08535663943c2f40d",

"supports_declared_by_raw_answer": "human-LLM factual-support agreement"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "page 6, Table 1",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "Claude Opus 4.5 ... 98.7% ... 95.7% ... 76.8%",

"quote_sha256": "8e88d02c36a4e4f27c1f74f8fee3442fce910fa93c790203cec48a745630b0c8",

"requested_url": "https://arxiv.org/abs/2605.06635",

"run_id": "V2-SOL-01",

"source_id": "s4_cited_not_verified",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "sp4a",

"span_occurrence_id": "V2-SOL-01:s4_cited_not_verified:sp4a",

"span_root_id": "span:sha256:9339db261929e3b56820214b791131dcbe342e20111e4b08916cfa523893810f",

"supports_declared_by_raw_answer": "coexistence of high link/relevance scores and lower factual support"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "pages 7-8, Tables 2-3",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "GPT-5.4 ... 78.6% ... 16.7% ... Claude Opus 4.6 ... 80.0% ... 57.9%",

"quote_sha256": "5fb4b4e6e9019214b7a5fef54027c65b1d397c3bd7c38fb58c960f3514697da5",

"requested_url": "https://arxiv.org/abs/2605.06635",

"run_id": "V2-SOL-01",

"source_id": "s4_cited_not_verified",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "sp4b",

"span_occurrence_id": "V2-SOL-01:s4_cited_not_verified:sp4b",

"span_root_id": "span:sha256:419fcae33b7758515a2334e64b7dca713a139a11d6930bc9cc772559989ae12e",

"supports_declared_by_raw_answer": "Fact Check endpoints for the 2-to-150-tool-call ablation"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "page 5, Section 3.3.3",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "contradicted, absent, or uncertain",

"quote_sha256": "e084f6ff65e9a116b127ab06d9745be6354fbcf3e66052c0182bc38177a9ae48",

"requested_url": "https://arxiv.org/abs/2605.06635",

"run_id": "V2-SOL-01",

"source_id": "s4_cited_not_verified",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "sp4c",

"span_occurrence_id": "V2-SOL-01:s4_cited_not_verified:sp4c",

"span_root_id": "span:sha256:c3dfc98566c27561eb7267972e00e3e1ba6a399a3ff5d002dca2d1233836790c",

"supports_declared_by_raw_answer": "conditions scored as Fact Check failure"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "page 3, Section 3.3",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "no archived snapshot exists at any timestamp",

"quote_sha256": "3a14d6f574bed17e76ef8b560bbca0aa27b3a50918e49461433e34ec7770a4c9",

"requested_url": "https://arxiv.org/abs/2604.03173",

"run_id": "V2-SOL-01",

"source_id": "s5_url_health",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "sp5a",

"span_occurrence_id": "V2-SOL-01:s5_url_health:sp5a",

"span_root_id": "span:sha256:a18e778fac3d49e81167f05e09fbc361e91af8c0c01b0fec52991f0a1e3d0b91",

"supports_declared_by_raw_answer": "operational definition of hallucinated URL"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "page 4, Table 2",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "openai-deepresearch ... 4,121 ... 10.1 ... 3.5 ... 6.6",

"quote_sha256": "501db42e628f70556263b3cb1961ea0a67c57fda022002ca72cbe9c7b8a608b3",

"requested_url": "https://arxiv.org/abs/2604.03173",

"run_id": "V2-SOL-01",

"source_id": "s5_url_health",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "sp5b",

"span_occurrence_id": "V2-SOL-01:s5_url_health:sp5b",

"span_root_id": "span:sha256:db9dc28183620369f5ad1fb0f179e8aadbc14c69c0fdb57b8a8b79cd8e95e2a6",

"supports_declared_by_raw_answer": "OpenAI Deep Research URL count and non-resolving/hallucinated/stale rates"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "page 4, Table 2",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "gemini-2.5-pro-deepres. ... 11,309 ... 18.5 ... 13.3 ... 5.2",

"quote_sha256": "63aba91235638fc3788940cd284864863aee605892d0334c712de8a591789029",

"requested_url": "https://arxiv.org/abs/2604.03173",

"run_id": "V2-SOL-01",

"source_id": "s5_url_health",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "sp5c",

"span_occurrence_id": "V2-SOL-01:s5_url_health:sp5c",

"span_root_id": "span:sha256:541aadd42c71d1bd9c2c458c8e250d5000ffb75f3801eea832a0a0ef58a3914b",

"supports_declared_by_raw_answer": "Gemini Deep Research URL count and non-resolving/hallucinated/stale rates"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 2, openai-deepresearch row",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "openai-deepresearch | OpenAI | 4,121 | 10.1 | 3.5 | 6.6",

"quote_sha256": "5bd6e5242ac0a8839758376181b806b9e9556b6c69e6a5b846c4bae5aaaf60cf",

"requested_url": "https://arxiv.org/html/2604.03173v1",

"run_id": "V2-SOL-02",

"source_id": "S1",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "S1_SPAN_1",

"span_occurrence_id": "V2-SOL-02:S1:S1_SPAN_1",

"span_root_id": "span:sha256:a7e8b6b9e8aa380a3d1eaedaae1439792bc858a48735284ecbd2822da1c3c0fe",

"supports_declared_by_raw_answer": "OpenAI Deep Research URL denominator and non-resolving, hallucinated, and stale percentages."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 2, gemini-2.5-pro-deepresearch row",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "gemini-2.5-pro-deepres. | Google | 11,309 | 18.5 | 13.3 | 5.2",

"quote_sha256": "66801673319516701b64b72ab382bc2924cafcd3a33d971598b22c8f90fa37bf",

"requested_url": "https://arxiv.org/html/2604.03173v1",

"run_id": "V2-SOL-02",

"source_id": "S1",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "S1_SPAN_2",

"span_occurrence_id": "V2-SOL-02:S1:S1_SPAN_2",

"span_root_id": "span:sha256:388e78d0c793138d79699c82540cb28cbd5722adfe6df119b36ee40a4502d9b5",

"supports_declared_by_raw_answer": "Gemini Deep Research URL denominator and non-resolving, hallucinated, and stale percentages."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "Main text, paragraph beginning 'For their high performances in generating reference lists'",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "citation-bearing sentences exhibited high error rates: 51.3 \u00b1 6.5% for ChatGPT and 57.8 \u00b1 22.7% for Le Chat",

"quote_sha256": "39f799b4bf1bcc50a74e5b019c585b970fa01779cf4b07bdeb27d71f35ea455b",

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"run_id": "V2-SOL-02",

"source_id": "S2",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "S2_SPAN_1",

"span_occurrence_id": "V2-SOL-02:S2:S2_SPAN_1",

"span_root_id": "span:sha256:092745c7db6407d4d52927642741cde229fad05a5347bb1867a5154421060e70",

"supports_declared_by_raw_answer": "Published mean sentence-level error rates across three runs."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-supplement-v2",

"license_treatment": "metadata and quote-minimal landing-page spans; file API required authentication and was not used",

"locator": "Supplementary Text 1, 'ChatGPT Deep Research: Results based on online search for evidence'",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "50.0 % in run #1, 45.5 % in run #2, and 58.3 % in run #3",

"quote_sha256": "d95bbfc885a7c2d9a88775f09d0428d6403d88a657b64ce142ae8b9366b400ce",

"requested_url": "https://data.mendeley.com/datasets/3s73z9zf3c/2",

"run_id": "V2-SOL-02",

"source_id": "S3",

"source_text_sha256": "2e96cc4ae8e28c60fcb94dd40ef64a15252e4416e7bf33185b8b76ea6ff228e4",

"source_work_id": "work:keplinger-supplement",

"span_id": "S3_SPAN_1",

"span_occurrence_id": "V2-SOL-02:S3:S3_SPAN_1",

"span_root_id": "span:sha256:870d5c8e029e0c3d25b0a4682848e6caafc22ee35db116517a55bf334eea3e2d",

"supports_declared_by_raw_answer": "Three manually audited ChatGPT Deep Research run rates; detailed tables give 8/16, 10/22, and 14/24."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-supplement-v2",

"license_treatment": "metadata and quote-minimal landing-page spans; file API required authentication and was not used",

"locator": "Supplementary Text 1, uploaded-papers mode summary",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "32.0% to 50.0% of citation-bearing sentences contained one or more",

"quote_sha256": "2cf619f684c7bbccc4e28b04f0b3f93f48e3f8e1289db1bce5b77d379b8bb001",

"requested_url": "https://data.mendeley.com/datasets/3s73z9zf3c/2",

"run_id": "V2-SOL-02",

"source_id": "S3",

"source_text_sha256": "2e96cc4ae8e28c60fcb94dd40ef64a15252e4416e7bf33185b8b76ea6ff228e4",

"source_work_id": "work:keplinger-supplement",

"span_id": "S3_SPAN_2",

"span_occurrence_id": "V2-SOL-02:S3:S3_SPAN_2",

"span_root_id": "span:sha256:3bbe9293af4ece39817697cf29b0fe4ab9423d429d9867dec2540b72c82fd5ba",

"supports_declared_by_raw_answer": "Uploaded-source restriction did not eliminate citation-bearing sentence inaccuracies."

},

{

"citation_carrier_accessible": false,

"edition_id": "edition:drbench-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Figure 1, FACT Citation Accuracy, final ICLR edition",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "FACT Citation Accuracy: 78.3, 75.0, 82.6, 73.1",

"quote_sha256": "dd39f211e8aa60e5851a5102b3ccc07f0549564f4d61b228a2dd89aa58bc6c40",

"requested_url": "https://openreview.net/pdf?id=hQ0K2Hhq7H",

"run_id": "V2-SOL-02",

"source_id": "S4",

"source_text_sha256": "ced9e7b0ae4ab2c26add3230a7b40200673ee4fc0c5c763df226bfe784d60016",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S4_SPAN_1",

"span_occurrence_id": "V2-SOL-02:S4:S4_SPAN_1",

"span_root_id": "span:sha256:c8683e7dad85f22158370acbb6a6e949b304b41bcba2c06f55cd31e90ce27130",

"supports_declared_by_raw_answer": "Final-edition citation-accuracy values in legend order: Gemini, OpenAI, Perplexity, Grok."

},

{

"citation_carrier_accessible": false,

"edition_id": "edition:drbench-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Appendix E, support judgment definition",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "support or not support",

"quote_sha256": "b940b72ea9971a1b22f3e492c50d3888e3c8090b1363dca86b4722f2234c9364",

"requested_url": "https://openreview.net/pdf?id=hQ0K2Hhq7H",

"run_id": "V2-SOL-02",

"source_id": "S4",

"source_text_sha256": "ced9e7b0ae4ab2c26add3230a7b40200673ee4fc0c5c763df226bfe784d60016",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S4_SPAN_2",

"span_occurrence_id": "V2-SOL-02:S4:S4_SPAN_2",

"span_root_id": "span:sha256:5899663b10cbe488c635c75743b9cf4ba1aab8c3e3a67604d4fa29248b387324",

"supports_declared_by_raw_answer": "Binary basis of each statement-URL judgment."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 1, OpenAI Deep Research row and cited-statements Match Rate column",

"match_status": "ordered-fragment-match",

"matched_fragments": [

"OpenAI Deep Research",

"Match Rate 78.87%"

],

"quote": "OpenAI Deep Research ... Match Rate 78.87%",

"quote_sha256": "9b1bf034e823976ac267434e9c67ee7e4038da04f3d2582a0de7d6d729fc6916",

"requested_url": "https://arxiv.org/html/2508.15804v1",

"run_id": "V2-SOL-02",

"source_id": "S5",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S5_SPAN_1",

"span_occurrence_id": "V2-SOL-02:S5:S5_SPAN_1",

"span_root_id": "span:sha256:e223640cb0ed12c87ca1d2406f3276f30a3b8d2017dd4dc1457fe59a94471040",

"supports_declared_by_raw_answer": "OpenAI citation-source semantic consistency."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 1, Gemini Deep Research row and cited-statements Match Rate column",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "Gemini Deep Research ... Match Rate 72.94%",

"quote_sha256": "460889dae2de871cd0538938b63dff194183b356e79fd6f944afb7d78eaebb32",

"requested_url": "https://arxiv.org/html/2508.15804v1",

"run_id": "V2-SOL-02",

"source_id": "S5",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S5_SPAN_2",

"span_occurrence_id": "V2-SOL-02:S5:S5_SPAN_2",

"span_root_id": "span:sha256:3b6f33fc8d800c8712c7b9fc2daafa5027f5176b23aa9120eeea323a21790a3a",

"supports_declared_by_raw_answer": "Gemini citation-source semantic consistency."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Official ICLR abstract",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "with citation accuracy ranging from 40\u201380% across systems",

"quote_sha256": "af76864a50c641f8d693216d3b7ebe2a414eebf75467305c1e1a6946b6746590",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/ad08767706825033b99122332293033d-Abstract-Conference.html",

"run_id": "V2-SOL-02",

"source_id": "S6",

"source_text_sha256": "0eaa415eca25bed5e591b8250640d695b86f6dac6601f5f00167ee3ac7ab9171",

"source_work_id": "work:deeptrace",

"span_id": "S6_SPAN_1",

"span_occurrence_id": "V2-SOL-02:S6:S6_SPAN_1",

"span_root_id": "span:sha256:a08468cb62f2feb21d9cbb95175ce00f54592bd83573cc9166e2375947f62fee",

"supports_declared_by_raw_answer": "Overall cross-system citation-accuracy range."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Paper section 3.1.1, factual-support validation",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "Pearson correlation of 0.62 between the LLM judge and manual labels",

"quote_sha256": "d2f6a4561a938ee913dadde2fc4a80c32c492fbcb3b47f236c9639375843ae8d",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/ad08767706825033b99122332293033d-Abstract-Conference.html",

"run_id": "V2-SOL-02",

"source_id": "S6",

"source_text_sha256": "0eaa415eca25bed5e591b8250640d695b86f6dac6601f5f00167ee3ac7ab9171",

"source_work_id": "work:deeptrace",

"span_id": "S6_SPAN_2",

"span_occurrence_id": "V2-SOL-02:S6:S6_SPAN_2",

"span_root_id": "span:sha256:e2c9f6220843a04665f5d1cd142ac15f04211ace5c579fa67f77e74e84a76dfe",

"supports_declared_by_raw_answer": "Moderate human agreement for the factual-support judge."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Abstract and Table 1",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "link validity above 94% and relevance above 80%, yet achieve only 39\u201377% factual accuracy",

"quote_sha256": "44d4c6941f2920e907de1e4cd22aeb9a725f24cc19432723754e316bd6f5cc49",

"requested_url": "https://arxiv.org/html/2605.06635v1",

"run_id": "V2-SOL-02",

"source_id": "S7",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S7_SPAN_1",

"span_occurrence_id": "V2-SOL-02:S7:S7_SPAN_1",

"span_root_id": "span:sha256:912c931063109f84ea88aea34891dd5e0f9147d2176af258059246431e9abfb1",

"supports_declared_by_raw_answer": "Separation between surface link quality and claim support."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 2, GPT-5.4 rows for 2 and 150 tool calls",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "2 | 100.0% | 100.0% | 78.6%; 150 | 99.2% | 99.2% | 16.7%",

"quote_sha256": "6d0377338a7fd91d8a513a277e9b64fd23420a14ef374f7f815ce45b88fc815c",

"requested_url": "https://arxiv.org/html/2605.06635v1",

"run_id": "V2-SOL-02",

"source_id": "S7",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S7_SPAN_2",

"span_occurrence_id": "V2-SOL-02:S7:S7_SPAN_2",

"span_root_id": "span:sha256:1f0504defcd0c58184e6958c7cf30ddd2120357feb41c2753301a5912457fef9",

"supports_declared_by_raw_answer": "Search-depth ablation endpoints for Link Works, Relevant Content, and Fact Check."

},

{

"citation_carrier_accessible": false,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "main text, paragraph beginning 'For their high performances'; Figure 1",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "citation-bearing sentences exhibited high error rates: 51.3 \u00b1 6.5% for ChatGPT and 57.8 \u00b1 22.7% for Le Chat",

"quote_sha256": "39f799b4bf1bcc50a74e5b019c585b970fa01779cf4b07bdeb27d71f35ea455b",

"requested_url": "https://onlinelibrary.wiley.com/doi/10.1111/jdv.70035",

"run_id": "V2-SOL-03",

"source_id": "s1",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "s1_span1",

"span_occurrence_id": "V2-SOL-03:s1:s1_span1",

"span_root_id": "span:sha256:46c392bc942cd88d525d74298e6bc6ebd9aeaec533ddfb7b7b761a9f5b454b05",

"supports_declared_by_raw_answer": "r2 and the distinction between identifiable references and actual claim support"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 1, OpenAI Deep Research row; FACT columns",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "OpenAI Deep Research | 46.98 | 46.87 | 45.25 | 49.27 | 47.14 | 77.96 | 40.79",

"quote_sha256": "dedd156999fa7d57ae9c22a166c56c263547eee3c37e6f41fe7644d980110986",

"requested_url": "https://arxiv.org/html/2506.11763",

"run_id": "V2-SOL-03",

"source_id": "s2",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "s2_span1",

"span_occurrence_id": "V2-SOL-03:s2:s2_span1",

"span_root_id": "span:sha256:2f311bfb451c6208101cb7583af37c338914fcdb10f5c75b131e23d5bb08e66a",

"supports_declared_by_raw_answer": "r3 OpenAI legacy citation accuracy and effective-citation values"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Table 1, Citation Metrics, column order GPT-5(DR), YouChat(DR), GPT-5(S), PPLX(DR), Copilot(TD), Gemini(DR)",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "%Citation Accuracy | 79.1 | 72.3 | 31.4 | 58.0 | 62.1 | 50.3",

"quote_sha256": "0a64ed8971e8457f4435d1334992e160d9e519df38aa242c8df0e17b15266532",

"requested_url": "https://arxiv.org/html/2509.04499",

"run_id": "V2-SOL-03",

"source_id": "s3",

"source_text_sha256": "d3d5e54884b46ff1815be6080405a9751e2b84b1a704a926e4af4806f68c9272",

"source_work_id": "work:deeptrace",

"span_id": "s3_span1",

"span_occurrence_id": "V2-SOL-03:s3:s3_span1",

"span_root_id": "span:sha256:91dd7d34a9546d71c8789056ebde06de25247281b1fb03ff41fc3a3ff20e9520",

"supports_declared_by_raw_answer": "r4 citation-accuracy values and configuration ordering"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 2, OpenAI Deep Research row",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "openai-deepresearch | OpenAI | 4,121 | 10.1 | 3.5 | 6.6",

"quote_sha256": "5bd6e5242ac0a8839758376181b806b9e9556b6c69e6a5b846c4bae5aaaf60cf",

"requested_url": "https://arxiv.org/html/2604.03173",

"run_id": "V2-SOL-03",

"source_id": "s4",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "s4_span1",

"span_occurrence_id": "V2-SOL-03:s4:s4_span1",

"span_root_id": "span:sha256:c130950377ad7bcb4218863dbbb72f86c43005d543056438e07501f32d369df7",

"supports_declared_by_raw_answer": "r5 OpenAI URL total, non-resolving, hallucinated, and stale rates"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 2, Gemini Deep Research row",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "gemini-2.5-pro-deepres. | Google | 11,309 | 18.5 | 13.3 | 5.2",

"quote_sha256": "66801673319516701b64b72ab382bc2924cafcd3a33d971598b22c8f90fa37bf",

"requested_url": "https://arxiv.org/html/2604.03173",

"run_id": "V2-SOL-03",

"source_id": "s4",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "s4_span2",

"span_occurrence_id": "V2-SOL-03:s4:s4_span2",

"span_root_id": "span:sha256:814248358095099e8d04475d7f17ce5aff8b4fe64f11239ea48edaf4d9a4b030",

"supports_declared_by_raw_answer": "r5 Gemini URL total, non-resolving, hallucinated, and stale rates"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 5.1 Results",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "the non-resolving rate drops significantly for all three models",

"quote_sha256": "7d1d86a3ca815ee38aafe98b3deb50c72abbf9f21f79336f8ab4843801b2dba5",

"requested_url": "https://arxiv.org/html/2604.03173",

"run_id": "V2-SOL-03",

"source_id": "s4",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "s4_span3",

"span_occurrence_id": "V2-SOL-03:s4:s4_span3",

"span_root_id": "span:sha256:b68153666e4e12aee8dc887fef0a61fa5f44b5f680a0b7eb95ff3defcf8b5bf1",

"supports_declared_by_raw_answer": "r6 intervention direction"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Abstract and Table 1",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "maintain link validity above 94% and relevance above 80%, yet achieve only 39\u201377% factual accuracy",

"quote_sha256": "e35ca66d12349d7290a6dbe7449bf6da173eb0671765644b581713528b316c74",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-SOL-03",

"source_id": "s5",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "s5_span1",

"span_occurrence_id": "V2-SOL-03:s5:s5_span1",

"span_root_id": "span:sha256:bb16d5f06abe8641d9ea6694e549aa15e240dd88946aa47a2d4e30ddd3351a24",

"supports_declared_by_raw_answer": "r7 surface-metric versus support gap"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 4.3, Tables 2\u20133",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "Fact Check accuracy drops approximately 42% on average from minimal (2 calls) to maximal search depth",

"quote_sha256": "4046b3ea7b246aa51b11f554509b50ebde3ca883cdcb8424e74a663b354bb845",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-SOL-03",

"source_id": "s5",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "s5_span2",

"span_occurrence_id": "V2-SOL-03:s5:s5_span2",

"span_root_id": "span:sha256:6e961498ed0f81a950f775956587f741e0eb9f11640fd3cb40870585558d869d",

"supports_declared_by_raw_answer": "r8 ablation summary"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:citation-verifier-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 1 contributions",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "624 attribution-citation pairs with gold labels for all 1,248 LLM-judged decisions, every one human-reviewed",

"quote_sha256": "8bebee03d81b15ad0af5b8ef69993f7e3cf525a7daa6e2608007fa9bac9b9a40",

"requested_url": "https://arxiv.org/abs/2607.08700",

"run_id": "V2-SOL-03",

"source_id": "s6",

"source_text_sha256": "0bd9a0bedc4b8ea9b08689109a9a6d763cb78d7513df2598874f147afa9ff9f7",

"source_work_id": "work:citation-verifier-benchmark",

"span_id": "s6_span1",

"span_occurrence_id": "V2-SOL-03:s6:s6_span1",

"span_root_id": "span:sha256:6885cf2524462022475b5829da081b2d35fcfe7463ca6ad0979e7c4fb1f0b54d",

"supports_declared_by_raw_answer": "r9 benchmark denominator and human-review basis"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-repo-main-469cce5",

"license_treatment": "metadata and quote-minimal README spans",

"locator": "README News, 2026-05-11 evaluator migration notice",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "Legacy code: the previous Gemini-2.5-Pro / Gemini-2.5-Flash evaluation code is preserved on the Gemini-2.5 branch.",

"quote_sha256": "02cd1c04d6312fcda76549885095d9c9d971cc14709ae8f749cf47fc2c11bcee",

"requested_url": "https://github.com/Ayanami0730/deep_research_bench",

"run_id": "V2-SOL-03",

"source_id": "s7",

"source_text_sha256": "afebe200cb45042b2783d6ad581770bbd8674027d4e7443e2eb6c26720c6cb0c",

"source_work_id": "work:deepresearch-bench-repository",

"span_id": "s7_span1",

"span_occurrence_id": "V2-SOL-03:s7:s7_span1",

"span_root_id": "span:sha256:e5bcf0241a1aeae7d9a92755b1bcf0adeeea93a8922f22f832693c61780386e7",

"supports_declared_by_raw_answer": "edition boundary for r3 and warning against mixing legacy and replacement evaluator scores"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Abstract, lines 47-48",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "yet achieve only 39\u201377% factual accuracy",

"quote_sha256": "be05510d58cc0231bf968f29c70b3d669a82d5cd0c74a778a5f8c4e035bf45b4",

"requested_url": "https://arxiv.org/html/2605.06635v1",

"run_id": "V2-SOL-04",

"source_id": "S1",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S1-SP1",

"span_occurrence_id": "V2-SOL-04:S1:S1-SP1",

"span_root_id": "span:sha256:74f6acafbb4c54d556a572460bf1b45bacd1e293f4aa16cc2d499174096b9c20",

"supports_declared_by_raw_answer": "Claim-support range despite high link validity and relevance."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 4.4, line 217",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "only 1 failed link out of 2,159 evaluations",

"quote_sha256": "da34a8e2a1d4e190d3b2095d79886b3cbf7edb2c63874189aa381302d2c2924e",

"requested_url": "https://arxiv.org/html/2605.06635v1",

"run_id": "V2-SOL-04",

"source_id": "S1",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S1-SP2",

"span_occurrence_id": "V2-SOL-04:S1:S1-SP2",

"span_root_id": "span:sha256:deada8bacb987a47f2a42e52fbfeb5c1beb3d250c0be79e2f9c5cdd6e16b4bc7",

"supports_declared_by_raw_answer": "Exact GPT-5.4 resolving-link numerator and denominator."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 4.3, line 212",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "from 79% to 17% (62%)",

"quote_sha256": "e29185e1dd93d69d1cf4328288773206e15b35990c7de068b31765e058bd50fd",

"requested_url": "https://arxiv.org/html/2605.06635v1",

"run_id": "V2-SOL-04",

"source_id": "S1",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S1-SP3",

"span_occurrence_id": "V2-SOL-04:S1:S1-SP3",

"span_root_id": "span:sha256:417182b4c15ac550469e435a02aab1e91b395f72f186312de9d34d57fd1c6c98",

"supports_declared_by_raw_answer": "Rounded GPT-5.4 Fact Check ablation change."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 1, Perplexity Deep Research citation-accuracy cell",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "90.24",

"quote_sha256": "64c911242980826f92ffae6d0571646ecc3aceca73c213791c6d3a8d889b4cc7",

"requested_url": "https://arxiv.org/html/2506.11763v1",

"run_id": "V2-SOL-04",

"source_id": "S2",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S2-SP1",

"span_occurrence_id": "V2-SOL-04:S2:S2-SP1",

"span_root_id": "span:sha256:196c54d6212fbc0c054dbf8e8467f2388c1f788bb93f4b1a4b6ae085eed17685",

"supports_declared_by_raw_answer": "Highest reported commercial-DRA citation accuracy."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 3.2, lines 144-146",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\u2018support\u2019 or \u2018not support\u2019",

"quote_sha256": "a49c200ea10833496ac73719992be09d5fd3fd021fdd51ccc587ac2592225f56",

"requested_url": "https://arxiv.org/html/2506.11763v1",

"run_id": "V2-SOL-04",

"source_id": "S2",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S2-SP2",

"span_occurrence_id": "V2-SOL-04:S2:S2-SP2",

"span_root_id": "span:sha256:13cb0ddb087b82988b7445a9cfb02562f9c044ef4f244e8b176ff475859eaea3",

"supports_declared_by_raw_answer": "Binary support-judgment definition."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Appendix C, line 365",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "96% of cases",

"quote_sha256": "ae845f85955fa47d52d51f32803e6a26afa8b725c523e53d9ef30b2abca72d9b",

"requested_url": "https://arxiv.org/html/2506.11763v1",

"run_id": "V2-SOL-04",

"source_id": "S2",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S2-SP3",

"span_occurrence_id": "V2-SOL-04:S2:S2-SP3",

"span_root_id": "span:sha256:b83996e3b5fdf878e04d6d41d0e7a1eebee9fdd9bbccf2ae4eaf33524fcfca44",

"supports_declared_by_raw_answer": "Reported judge-human alignment on support determinations."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Appendix C, line 365",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "92% of cases",

"quote_sha256": "e9c779296734583861b22f45da62bff61c9c899018602fda17716d69fc449a63",

"requested_url": "https://arxiv.org/html/2506.11763v1",

"run_id": "V2-SOL-04",

"source_id": "S2",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S2-SP4",

"span_occurrence_id": "V2-SOL-04:S2:S2-SP4",

"span_root_id": "span:sha256:f35c9565f2fd047f6e7272dc608bbf6a63813a717bad5354c24bdaac65e40bfe",

"supports_declared_by_raw_answer": "Reported judge-human alignment on not-support determinations."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 3.3, line 155",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "78.87% vs. 72.94%",

"quote_sha256": "4545fdd46cab2ffc151c9ddbfaab08b82862b68eebf5205cda180b5875a3aa81",

"requested_url": "https://arxiv.org/html/2508.15804v1",

"run_id": "V2-SOL-04",

"source_id": "S3",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S3-SP1",

"span_occurrence_id": "V2-SOL-04:S3:S3-SP1",

"span_root_id": "span:sha256:6c1d327e58ef001533ea1c11010e974436bbeafbde138efcc7b516281f6d25a5",

"supports_declared_by_raw_answer": "OpenAI versus Gemini citation-match comparison."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 4, line 186",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "the cited URL does not exist",

"quote_sha256": "ac4714fdd34ead02999923441a133a798a67254f0449f293905ef5b1d1caffab",

"requested_url": "https://arxiv.org/html/2508.15804v1",

"run_id": "V2-SOL-04",

"source_id": "S3",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S3-SP2",

"span_occurrence_id": "V2-SOL-04:S3:S3-SP2",

"span_root_id": "span:sha256:3015a109e7898cb720e426e9a57c526277b648c75f1fb4b13631b00524eac943",

"supports_declared_by_raw_answer": "Documented non-resolving fabricated-link failure."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Abstract, line 46",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "citation accuracy ranging from 40\u201380% across systems",

"quote_sha256": "a0eeca5c86f9c7aeda9167f626a8cb0ed56dc87ea1ac5ca8c4a7a0add76e7625",

"requested_url": "https://arxiv.org/html/2509.04499v1",

"run_id": "V2-SOL-04",

"source_id": "S4",

"source_text_sha256": "d3d5e54884b46ff1815be6080405a9751e2b84b1a704a926e4af4806f68c9272",

"source_work_id": "work:deeptrace",

"span_id": "S4-SP1",

"span_occurrence_id": "V2-SOL-04:S4:S4-SP1",

"span_root_id": "span:sha256:8cce3d9ee1e1674ba1b93321dd9b99a55d834ca980451f8ab61bcaf5bdda8e4d",

"supports_declared_by_raw_answer": "Reported cross-system citation-support range."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Section 3.2, lines 163-171",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "303 queries x 9 models",

"quote_sha256": "5cc83ab699cb1342475b407e34d33fc106d64b5d9270e5bb9445d80586ebabc2",

"requested_url": "https://arxiv.org/html/2509.04499v1",

"run_id": "V2-SOL-04",

"source_id": "S4",

"source_text_sha256": "d3d5e54884b46ff1815be6080405a9751e2b84b1a704a926e4af4806f68c9272",

"source_work_id": "work:deeptrace",

"span_id": "S4-SP2",

"span_occurrence_id": "V2-SOL-04:S4:S4-SP2",

"span_root_id": "span:sha256:9dd6eb46529e77a62748093528e52685fbd92d46a756aef750d22aa25981db23",

"supports_declared_by_raw_answer": "Response-sample basis."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "Table 1, ChatGPT subtotal",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "23 (100.0%)",

"quote_sha256": "5428bd1a04d5d2e332015107e915fe3a38f1d5e173e8c7246b743d98ecfd8618",

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"run_id": "V2-SOL-04",

"source_id": "S5",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "S5-SP1",

"span_occurrence_id": "V2-SOL-04:S5:S5-SP1",

"span_root_id": "span:sha256:8c8d9a6dc003bf0298542b85df3f6f08db631e68e89f606f4e53e36e86dde167",

"supports_declared_by_raw_answer": "ChatGPT reference denominator."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "Table 1, ChatGPT all-correct subtotal",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "16 (69.6%)",

"quote_sha256": "b6c3ff6f2ff60f37ae1c6b773ba6aa5a6c01015b61e3cc6f2a06bfcef69b53ad",

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"run_id": "V2-SOL-04",

"source_id": "S5",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "S5-SP2",

"span_occurrence_id": "V2-SOL-04:S5:S5-SP2",

"span_root_id": "span:sha256:39a9931f05ea356fe904ed15768fa08c66440b1fcb20f98e267a315e6d562af9",

"supports_declared_by_raw_answer": "Entirely correct ChatGPT references."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "Paragraph after Table 1",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "51.3 \u00b1 6.5%",

"quote_sha256": "1f863c1f35c37af1fd2c3e4927ea0b289d94a9532fa5855e209ce98ea344d498",

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"run_id": "V2-SOL-04",

"source_id": "S5",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "S5-SP3",

"span_occurrence_id": "V2-SOL-04:S5:S5-SP3",

"span_root_id": "span:sha256:f05b3189dc3fafd45b5bde643a91fd0616832928215a37a2430a5a658e592029",

"supports_declared_by_raw_answer": "ChatGPT citation-bearing-sentence error rate."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-supplement-v1",

"license_treatment": "metadata and quote-minimal landing-page spans",

"locator": "Dataset description",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "sentence-level evaluation",

"quote_sha256": "e3ec5c81f615c657f7c6b124c5b2150d86f18281384ca0bd24ebb43a3317796f",

"requested_url": "https://data.mendeley.com/datasets/3s73z9zf3c/1",

"run_id": "V2-SOL-04",

"source_id": "S6",

"source_text_sha256": "1ed2ccfeb91e0346787a5c3122e191a6798d522a86f740e10ae9956c5b3f4cd4",

"source_work_id": "work:keplinger-supplement",

"span_id": "S6-SP1",

"span_occurrence_id": "V2-SOL-04:S6:S6-SP1",

"span_root_id": "span:sha256:94ad583514aa7e04b8d5f7b8f9cdda6fdffe8ca6712f033da3ce4f18fe7c03f1",

"supports_declared_by_raw_answer": "Unit of claim-citation analysis."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:citation-verifier-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Abstract, line 16",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "1,248 rubric decisions",

"quote_sha256": "0e980e624cddfead247a93771e68f3c678bd30495cd310fd81c0812ecab2bb8b",

"requested_url": "https://arxiv.org/abs/2607.08700",

"run_id": "V2-SOL-04",

"source_id": "S7",

"source_text_sha256": "0bd9a0bedc4b8ea9b08689109a9a6d763cb78d7513df2598874f147afa9ff9f7",

"source_work_id": "work:citation-verifier-benchmark",

"span_id": "S7-SP1",

"span_occurrence_id": "V2-SOL-04:S7:S7-SP1",

"span_root_id": "span:sha256:e54ee1457854a9a5e46f7c7c2e66ac6a5a69b963f8ee3450f40052f138d5d0c1",

"supports_declared_by_raw_answer": "Verifier-benchmark denominator."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:citation-verifier-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Abstract, line 16",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "378 of which were hard cases",

"quote_sha256": "6a7d5dd08b03dbb3fb7acaa3d8f8943e1ee7d8dcaeb8b2b829ba36a7963e4991",

"requested_url": "https://arxiv.org/abs/2607.08700",

"run_id": "V2-SOL-04",

"source_id": "S7",

"source_text_sha256": "0bd9a0bedc4b8ea9b08689109a9a6d763cb78d7513df2598874f147afa9ff9f7",

"source_work_id": "work:citation-verifier-benchmark",

"span_id": "S7-SP2",

"span_occurrence_id": "V2-SOL-04:S7:S7-SP2",

"span_root_id": "span:sha256:794bdf7e0595799afbb0d55f3dc522077aed9615c59f1e3575707148cf444bcf",

"supports_declared_by_raw_answer": "Human-adjudicated difficult-case basis."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "p. 4, lines 150-159",

"match_status": "ordered-fragment-match",

"matched_fragments": [

"Each unique Statement-URL pair undergoes a support evaluation",

"This results in a binary judgment ('support' or 'not support') for each pair, determining whether the citation accurately grounds the claim"

],

"quote": "\"Each unique Statement-URL pair undergoes a support evaluation... This results in a binary judgment ('support' or 'not support') for each pair, determining whether the citation accurately grounds the claim.\"",

"quote_sha256": "4edb75ec3f690eb2d15089923ba5e9c4e6174f55094d0919041bf10625901ceb",

"requested_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"run_id": "V2-TERRA-01",

"source_id": "s1_deepresearchbench",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "s1_method",

"span_occurrence_id": "V2-TERRA-01:s1_deepresearchbench:s1_method",

"span_root_id": "span:sha256:b0a39c963cdfdc5f9852cfc9f7e33e05d31232bd1d8998851e879c38e6c34843",

"supports_declared_by_raw_answer": "FACT directly evaluates claim-to-citation support."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "p. 5, Table 1, lines 179-198",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"Perplexity Deep Research ... 90.24 31.26\"; \"Gemini-2.5-Pro Deep Research ... 81.44 111.21\"; \"OpenAI Deep Research ... 77.96 40.79\".",

"quote_sha256": "b6e43bff5e4d384a69288ea69405d3ff49f44c00c5a8b6ab44e3a5a427b8bee1",

"requested_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"run_id": "V2-TERRA-01",

"source_id": "s1_deepresearchbench",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "s1_table",

"span_occurrence_id": "V2-TERRA-01:s1_deepresearchbench:s1_table",

"span_root_id": "span:sha256:77001af3bf7a976754d0b5d149da1c4e9b464c3515694958fe01a3c4ca8c7a5c",

"supports_declared_by_raw_answer": "Reported FACT Citation Accuracy and Effective Citations values."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "p. 17, Table 5, lines 715-729",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"OpenAI Deep Research April 1 \u2013 May 8\"; \"Gemini 2.5 Pro Deep Research April 27 \u2013 April 29\"; \"Perplexity Deep Research April 1 \u2013 April 29\".",

"quote_sha256": "bc8f24337ac0718946ee4dfb90675a92120f890d5ca0d25fdc4279c368a1875e",

"requested_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"run_id": "V2-TERRA-01",

"source_id": "s1_deepresearchbench",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "s1_dates",

"span_occurrence_id": "V2-TERRA-01:s1_deepresearchbench:s1_dates",

"span_root_id": "span:sha256:34a61ece593714d9608073ee5aeea22635a3fb111582fc8e2e8bf0b371955b70",

"supports_declared_by_raw_answer": "Output-vintage scope."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "p. 5, lines 289-300",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"we retrieve the full content of each cited webpage via web scraping... [and] perform consistency verification by comparing the statement with the retrieved content\".",

"quote_sha256": "191b5aedee7e0a6a8d59f0be1c67a99f37326461a0baadabcc66d181f917d6d1",

"requested_url": "https://arxiv.org/pdf/2508.15804",

"run_id": "V2-TERRA-01",

"source_id": "s2_reportbench",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "s2_method",

"span_occurrence_id": "V2-TERRA-01:s2_reportbench:s2_method",

"span_root_id": "span:sha256:f4c499ea114f2647a0dc4830b72e4b9120d6c368f846efb1ccd3ae4c0ed4f2c7",

"supports_declared_by_raw_answer": "Citation-claim consistency method."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "p. 6, Table 1, lines 333-355",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"For cited statements, we compute the match rate, i.e., the proportion of statements that are semantically consistent with their cited sources.\"; \"OpenAI Deep Research ... 78.87% 88.2\"; \"Gemini Deep Research ... 72.94% 96.2\".",

"quote_sha256": "427d50b4f83bb70f9b686fc45eb7b484eb0587caab4658d55301a555b371d7e5",

"requested_url": "https://arxiv.org/pdf/2508.15804",

"run_id": "V2-TERRA-01",

"source_id": "s2_reportbench",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "s2_metric_table",

"span_occurrence_id": "V2-TERRA-01:s2_reportbench:s2_metric_table",

"span_root_id": "span:sha256:a6bde87f202e383e8bdd8403b2e5241b65380e0616e61b18f36a7c3719b55d76",

"supports_declared_by_raw_answer": "Metric definition and values."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "p. 5, lines 315-329",

"match_status": "ordered-fragment-match",

"matched_fragments": [

"we manually collected responses",

"during the period from July 14 to July 25",

"OpenAI was using the standard version of Deep Research, powered by the o3 model"

],

"quote": "\"we manually collected responses ... during the period from July 14 to July 25... OpenAI was using the standard version of Deep Research, powered by the o3 model.\"",

"quote_sha256": "e45fbf61feb38fc8de32433d439fdb8544133e905103b884c92327930a929b0a",

"requested_url": "https://arxiv.org/pdf/2508.15804",

"run_id": "V2-TERRA-01",

"source_id": "s2_reportbench",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "s2_collection",

"span_occurrence_id": "V2-TERRA-01:s2_reportbench:s2_collection",

"span_root_id": "span:sha256:7a6046a756d6ec0315d330cf0817d062ea48498bdc9c90b3489d64b2bd12a3a4",

"supports_declared_by_raw_answer": "Agent configuration and collection period."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:researcherbench-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "p. 6, lines 361-387",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"Faithfulness score... evaluates the proportion of cited claims that are actually supported by their referenced sources\"; \"Groundedness score... measures the proportion of all factual claims that have explicit citation support\".",

"quote_sha256": "2e2d475bdbdccacc310754eb68bb8f1b5dac25480684dc29773cbb8cf9141ea0",

"requested_url": "https://arxiv.org/pdf/2507.16280",

"run_id": "V2-TERRA-01",

"source_id": "s3_researcherbench",

"source_text_sha256": "1aa77342c2571bb564fe1b3e9a2ba2333b4ef0d637c4a3ef622540741927b0ec",

"source_work_id": "work:researcherbench",

"span_id": "s3_definition",

"span_occurrence_id": "V2-TERRA-01:s3_researcherbench:s3_definition",

"span_root_id": "span:sha256:5351ba5f0c5afba376e1a70ffafab3d6a4eefae1d9507b5206a64951f360a095",

"supports_declared_by_raw_answer": "Support and coverage definitions."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:researcherbench-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "p. 6-7, lines 395-409",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"We evaluated several leading commercial deep research systems\" including OpenAI, Gemini, Grok, and Perplexity; \"All evaluations were conducted between March and April in 2025\".",

"quote_sha256": "bf48819f2f28fb5eeb1fde9568c7398687840f03034e87f0f634df8416b91d64",

"requested_url": "https://arxiv.org/pdf/2507.16280",

"run_id": "V2-TERRA-01",

"source_id": "s3_researcherbench",

"source_text_sha256": "1aa77342c2571bb564fe1b3e9a2ba2333b4ef0d637c4a3ef622540741927b0ec",

"source_work_id": "work:researcherbench",

"span_id": "s3_models_time",

"span_occurrence_id": "V2-TERRA-01:s3_researcherbench:s3_models_time",

"span_root_id": "span:sha256:5fb35a076c774fd7141fd642f5490f3679fc2452e9e886b62d2e3c370b9c029d",

"supports_declared_by_raw_answer": "System and time scope."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:researcherbench-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "p. 7, Table 2, lines 415-427",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"OpenAI Deep Research 0.7032 0.84 0.34\"; \"Gemini Deep Research 0.6929 0.86 0.59\"; \"Perplexity Deep Research 0.4800 0.85 0.56\".",

"quote_sha256": "bf2cdd33d00d7b1110c4ac5d5af27d53a7acc505bf17ea7796e029bdf31c948f",

"requested_url": "https://arxiv.org/pdf/2507.16280",

"run_id": "V2-TERRA-01",

"source_id": "s3_researcherbench",

"source_text_sha256": "1aa77342c2571bb564fe1b3e9a2ba2333b4ef0d637c4a3ef622540741927b0ec",

"source_work_id": "work:researcherbench",

"span_id": "s3_table",

"span_occurrence_id": "V2-TERRA-01:s3_researcherbench:s3_table",

"span_root_id": "span:sha256:eadcf02c45f4b152c73c6f871d116e7d1db8e78c796671580a5eb72e7ff16687",

"supports_declared_by_raw_answer": "Faithfulness and groundedness results."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "pp. 5-6, lines 209-217 and Table 1, lines 254-271",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"Fact Check verifies whether specific factual claims are accurately supported by the source content\"; \"Claude Opus 4.5 ... 98.7% 95.7% 76.8%\"; \"GPT-5 Mini ... 99.3% 87.4% 38.9%\".",

"quote_sha256": "609cbee79dcf2e423af2abff67454fa2c88b3c30ae286fd7f8c8f88ed2ddd73f",

"requested_url": "https://arxiv.org/pdf/2605.06635",

"run_id": "V2-TERRA-01",

"source_id": "s4_cited_not_verified",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "s4_definition_table",

"span_occurrence_id": "V2-TERRA-01:s4_cited_not_verified:s4_definition_table",

"span_root_id": "span:sha256:48a692e799a44e37783d312cb80a0500cca0aa94abb30dc074c1cb6f0bb0ead1",

"supports_declared_by_raw_answer": "Support metric, evaluation protocol, and model results."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "pp. 6-7, lines 286-318",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"Fact Check accuracy drops approximately 42% on average\"; \"GPT-5.4 ... 78.6%\" at 2 calls and \"16.7%\" at 150; Claude Opus 4.6 \"80.0%\" at 2 and \"57.9%\" at 150.",

"quote_sha256": "b75318c065718db1259a86220f8793c81ef12ee9798d8fc4e0d4354fd83c2e45",

"requested_url": "https://arxiv.org/pdf/2605.06635",

"run_id": "V2-TERRA-01",

"source_id": "s4_cited_not_verified",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "s4_depth",

"span_occurrence_id": "V2-TERRA-01:s4_cited_not_verified:s4_depth",

"span_root_id": "span:sha256:a9625285d829a0463b566e89d328e49b9c21cc07e83dbf845ac422f5a8e4b72e",

"supports_declared_by_raw_answer": "Search-depth ablation."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "pp. 1-2, lines 64-95",

"match_status": "ordered-fragment-match",

"matched_fragments": [

"This study focuses on URL-based citation hallucinations",

"fabricated snippets",

"and invented bibliographic entries",

"require separate systematic study"

],

"quote": "\"This study focuses on URL-based citation hallucinations\"; \"fabricated snippets ... and invented bibliographic entries ... require separate systematic study.\"",

"quote_sha256": "ee1de7374c67e208aa5d1331c20a08c8c83418309f1bd768d380620bfdc27901",

"requested_url": "https://arxiv.org/pdf/2604.03173",

"run_id": "V2-TERRA-01",

"source_id": "s5_urlhealth",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "s5_scope",

"span_occurrence_id": "V2-TERRA-01:s5_urlhealth:s5_scope",

"span_root_id": "span:sha256:55de6a6ca9414a252592e9f77544a9259533dd713ce1f12f6a216752a2b76e8f",

"supports_declared_by_raw_answer": "The study does not measure semantic support."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "p. 3, Table 2, lines 173-200",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"openai-deepresearch OpenAI 4,121 10.1 ... 3.5 ... 6.6\"; \"gemini-2.5-pro-deepres. Google 11,309 18.5 ... 13.3 ... 5.2\".",

"quote_sha256": "83d55ca96f5207144829514b8d88e689901701eb1f4b992dd6b2e5af8c288924",

"requested_url": "https://arxiv.org/pdf/2604.03173",

"run_id": "V2-TERRA-01",

"source_id": "s5_urlhealth",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "s5_table",

"span_occurrence_id": "V2-TERRA-01:s5_urlhealth:s5_table",

"span_root_id": "span:sha256:6af594d0155420f020802bee4a4dae0cd19f46d2c69dc107c69e75e298299169",

"supports_declared_by_raw_answer": "Per-model resolution and Wayback-based classification rates."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "p. 4, lines 210-226",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"Pooling across the two deep research agents, the hallucination rate is 10.7% ... versus 4.8% ... for the eight search-augmented models\"; \"Non-resolving rates ... 16.2% ... versus 6.8%\".",

"quote_sha256": "923c3a4f90999d5c32b7eacfcabb033566f20427fbe4078c2404eb3725ae7226",

"requested_url": "https://arxiv.org/pdf/2604.03173",

"run_id": "V2-TERRA-01",

"source_id": "s5_urlhealth",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "s5_comparison",

"span_occurrence_id": "V2-TERRA-01:s5_urlhealth:s5_comparison",

"span_root_id": "span:sha256:baa7644423c27da0263b753522f1c14915dfdb8f9505577f8ff755a0e668c023",

"supports_declared_by_raw_answer": "Pooled comparison and statistical basis."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "p.4, Section 3.2, lines 148-155",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"Each unique Statement-URL pair undergoes a support evaluation.\"",

"quote_sha256": "3cca518ed14d7451346a35623b20e7f9e350e4cb5cbcec3d7f622cc6aac509ce",

"requested_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"run_id": "V2-TERRA-02",

"source_id": "s1_deepresearchbench",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "s1_fact_method",

"span_occurrence_id": "V2-TERRA-02:s1_deepresearchbench:s1_fact_method",

"span_root_id": "span:sha256:eaf37bef7b8dc35f3e00b85d3e58ad872dc0ea604a062c8cdd8929184587d5d9",

"supports_declared_by_raw_answer": "FACT extracts, deduplicates, fetches webpage text, and judges binary claim support."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "p.5, Table 1, Deep Research Agent rows",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"Perplexity Deep Research ... 90.24 31.26\"",

"quote_sha256": "c5b9745d64c8a1f99fcbbc08a6faebbf6cea6a3aac271effa2dbe31efb82d0eb",

"requested_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"run_id": "V2-TERRA-02",

"source_id": "s1_deepresearchbench",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "s1_fact_results",

"span_occurrence_id": "V2-TERRA-02:s1_deepresearchbench:s1_fact_results",

"span_root_id": "span:sha256:b74af48901531355cd4351211eac41aef691ea4eb29488347782fca1ec2d1eb8",

"supports_declared_by_raw_answer": "The Table 1 citation-accuracy and effective-citation values; the other three DRA rows are in the same table."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "pp.17-18, Appendix E, equations 4-6",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"the proportion of 'support' statement-URL pairs for each individual task\"",

"quote_sha256": "82b9361b6e696a92b41fdfae500f56b05d48575c0d429f15b3a8d6786c83b8d7",

"requested_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"run_id": "V2-TERRA-02",

"source_id": "s1_deepresearchbench",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "s1_fact_formula",

"span_occurrence_id": "V2-TERRA-02:s1_deepresearchbench:s1_fact_formula",

"span_root_id": "span:sha256:e4a9f30effc031388ed5c9993dc6f071903bec5e3f1998db4c08ab00b284b27f",

"supports_declared_by_raw_answer": "Macro-average citation-accuracy definition and effective-citation calculation."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "p.17, Appendix C, lines 704-708",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"aligned with human 'support' determinations in 96% of cases\"",

"quote_sha256": "43c892b582d42a335c982009f97347b5bf2ae2799cc3e478c53504f4d01e4388",

"requested_url": "https://deepresearch-bench.github.io/static/papers/deepresearch-bench.pdf",

"run_id": "V2-TERRA-02",

"source_id": "s1_deepresearchbench",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "s1_fact_judge_validation",

"span_occurrence_id": "V2-TERRA-02:s1_deepresearchbench:s1_fact_judge_validation",

"span_root_id": "span:sha256:f98f1274859ba98cc538e3be6d11d345a5382cfcb9516aa1cf3f3278ea646b92",

"supports_declared_by_raw_answer": "Reported 100-pair judge-validation result, including 92% agreement on not-support determinations."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:liveresearchbench-arxiv-v5",

"license_treatment": "quote-minimal attributed spans; raw CC BY claim is corrected in review records",

"locator": "p.33, Appendix E",

"match_status": "ordered-fragment-match",

"matched_fragments": [

"E1: URL is inaccessible or does not resolve; E2: URL content is irrelevant",

"E3: URL content does not support the specific statements"

],

"quote": "\"E1: URL is inaccessible or does not resolve; E2: URL content is irrelevant ... E3: URL content does not support the specific statements.\"",

"quote_sha256": "064c8e109876030c9fdc887811e31b1e5505c4bb3968369b65b26566813cca3e",

"requested_url": "https://arxiv.org/pdf/2510.14240v5",

"run_id": "V2-TERRA-02",

"source_id": "s2_liveresearchbench",

"source_text_sha256": "6605fcf67de305d0313f395e2493bf6c3bbd7ae87d7fa08358364ded225d9b77",

"source_work_id": "work:liveresearchbench",

"span_id": "s2_rubric_tree",

"span_occurrence_id": "V2-TERRA-02:s2_liveresearchbench:s2_rubric_tree",

"span_root_id": "span:sha256:d880991ad784390029d49f84949d083c4789d67d45893d87cfb6b5a38703453f",

"supports_declared_by_raw_answer": "Exact failure-class definitions and the resolving/relevance/support verification sequence."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:liveresearchbench-arxiv-v5",

"license_treatment": "quote-minimal attributed spans; raw CC BY claim is corrected in review records",

"locator": "p.33, Table 7",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"GPT-5 4.2 1.7 13.3 19.2\"",

"quote_sha256": "9947eb1b16a73e4f266a2d2cd0cfcbbe40f403428bab87f653e3634580182437",

"requested_url": "https://arxiv.org/pdf/2510.14240v5",

"run_id": "V2-TERRA-02",

"source_id": "s2_liveresearchbench",

"source_text_sha256": "6605fcf67de305d0313f395e2493bf6c3bbd7ae87d7fa08358364ded225d9b77",

"source_work_id": "work:liveresearchbench",

"span_id": "s2_table7",

"span_occurrence_id": "V2-TERRA-02:s2_liveresearchbench:s2_table7",

"span_root_id": "span:sha256:256817dca0b5aac5741ec51862ec8d0f9b4b93ea78186a4041052c7b2d1c4382",

"supports_declared_by_raw_answer": "Wide Info Search error counts; Table 7 also supplies the Grok-4, Open Deep Research, and Market Analysis rows."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:liveresearchbench-arxiv-v5",

"license_treatment": "quote-minimal attributed spans; raw CC BY claim is corrected in review records",

"locator": "p.23, Appendix C, Citation Accuracy",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"human evaluators agree with either Gemini or GPT-5 on 87.1%\"",

"quote_sha256": "0d570d54a0712207d83f71aae02fc0c8ca424a120d61b36c0d0e243611d8f842",

"requested_url": "https://arxiv.org/pdf/2510.14240v5",

"run_id": "V2-TERRA-02",

"source_id": "s2_liveresearchbench",

"source_text_sha256": "6605fcf67de305d0313f395e2493bf6c3bbd7ae87d7fa08358364ded225d9b77",

"source_work_id": "work:liveresearchbench",

"span_id": "s2_validation",

"span_occurrence_id": "V2-TERRA-02:s2_liveresearchbench:s2_validation",

"span_root_id": "span:sha256:e6f447129d9a662fb202d4ac22e16822bacfdbcbcdb7afe3b589e21802890b8a",

"supports_declared_by_raw_answer": "Reported human agreement on 200 sampled claim--URL support judgments."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "p.6, Section 3.1.4, equation 7",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"fraction of statement citations that accurately reflect that a source's content supports the statement\"",

"quote_sha256": "b08cab9336fb2681d652b0f8d512cbf8d45ecf977ed5130269b5e909d96435e7",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"run_id": "V2-TERRA-02",

"source_id": "s3_deeptrace",

"source_text_sha256": "0eaa415eca25bed5e591b8250640d695b86f6dac6601f5f00167ee3ac7ab9171",

"source_work_id": "work:deeptrace",

"span_id": "s3_definition",

"span_occurrence_id": "V2-TERRA-02:s3_deeptrace:s3_definition",

"span_root_id": "span:sha256:4d92f4721bb3d81403154a23d7fde5b342251d50a2d1034ec6d4542a35e9fc8b",

"supports_declared_by_raw_answer": "Citation-accuracy metric definition."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "p.8, Table 1",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"%Citation Accuracy 79.1 ... 39.33 ... 72.3 ... 58.0 ... 62.1 ... 50.3\"",

"quote_sha256": "015ad79fdb7279dd5f90e40459c389178dd660327dfd04ceb74ea002814289eb",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"run_id": "V2-TERRA-02",

"source_id": "s3_deeptrace",

"source_text_sha256": "0eaa415eca25bed5e591b8250640d695b86f6dac6601f5f00167ee3ac7ab9171",

"source_work_id": "work:deeptrace",

"span_id": "s3_table1",

"span_occurrence_id": "V2-TERRA-02:s3_deeptrace:s3_table1",

"span_root_id": "span:sha256:7572651679e7e8a0a1a6e394504e75749ae2649cfb55f7a80531c8b53acb31dc",

"supports_declared_by_raw_answer": "Reported DR-system citation-accuracy values and adjacent unsupported-statement/thoroughness rows."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "pp.4-7, Sections 3.1.1 and 3.2",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"The dataset comprises 303 questions\"",

"quote_sha256": "2ce91759797b86455a478a9ced58ec59d032e1a32a1c964347f96be4ab9e2054",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"run_id": "V2-TERRA-02",

"source_id": "s3_deeptrace",

"source_text_sha256": "0eaa415eca25bed5e591b8250640d695b86f6dac6601f5f00167ee3ac7ab9171",

"source_work_id": "work:deeptrace",

"span_id": "s3_corpus_and_retrieval",

"span_occurrence_id": "V2-TERRA-02:s3_deeptrace:s3_corpus_and_retrieval",

"span_root_id": "span:sha256:9cb1700c08368feb234207045354cbf559d6ed80f2b0bae814d2b5b3c858b73c",

"supports_declared_by_raw_answer": "Population, public-UI collection, Jina Reader path, and stated source-extraction exclusions."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:deeptrace-iclr-2026",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "p.14, Table 3",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"Factual support (statement\u2013source) 0.62\"",

"quote_sha256": "75a9212dc448c15ffc1e4a36bdf190728977470661bd9f45be534010704faa25",

"requested_url": "https://proceedings.iclr.cc/paper_files/paper/2026/file/ad08767706825033b99122332293033d-Paper-Conference.pdf",

"run_id": "V2-TERRA-02",

"source_id": "s3_deeptrace",

"source_text_sha256": "0eaa415eca25bed5e591b8250640d695b86f6dac6601f5f00167ee3ac7ab9171",

"source_work_id": "work:deeptrace",

"span_id": "s3_judge_validation",

"span_occurrence_id": "V2-TERRA-02:s3_deeptrace:s3_judge_validation",

"span_root_id": "span:sha256:aa1534ce2e177153ab85e512deb8d064d14ec1eae16d8032bdb5043252738d4c",

"supports_declared_by_raw_answer": "Reported Pearson correlation between LLM factual-support labels and human annotations, N=100 per task."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:researcherbench-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Section 4.2.1, Citation Support Verification and Score Computation",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"whether the extracted content supports the corresponding claim\"",

"quote_sha256": "12525ade29e741070aec9df092b69ee828e7f4333a4c41c331784cfab48bef56",

"requested_url": "https://arxiv.org/html/2507.16280",

"run_id": "V2-TERRA-02",

"source_id": "s4_researcherbench",

"source_text_sha256": "1aa77342c2571bb564fe1b3e9a2ba2333b4ef0d637c4a3ef622540741927b0ec",

"source_work_id": "work:researcherbench",

"span_id": "s4_method",

"span_occurrence_id": "V2-TERRA-02:s4_researcherbench:s4_method",

"span_root_id": "span:sha256:886b50d94e3facabc487342cd2af5157cf956622ecfc368684596c5f6b61c350",

"supports_declared_by_raw_answer": "Claim--URL--context extraction, Jina Reader retrieval, binary support decision, and faithfulness/groundedness definitions."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:researcherbench-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Section 5.2, Table 2",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"OpenAI Deep Research | 0.7032 | 0.84 | 0.34\"",

"quote_sha256": "5c9ca63ca1d8757a3e9e598d441fbca1675657a026bb4f97424112eded86594d",

"requested_url": "https://arxiv.org/html/2507.16280",

"run_id": "V2-TERRA-02",

"source_id": "s4_researcherbench",

"source_text_sha256": "1aa77342c2571bb564fe1b3e9a2ba2333b4ef0d637c4a3ef622540741927b0ec",

"source_work_id": "work:researcherbench",

"span_id": "s4_results",

"span_occurrence_id": "V2-TERRA-02:s4_researcherbench:s4_results",

"span_root_id": "span:sha256:f85edb1051cbc19ee38bc796ed3392abf5e44fc869d59d7e82bfe2cb81696e9b",

"supports_declared_by_raw_answer": "Coverage, faithfulness, and groundedness values; the other systems appear in the same table."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Abstract, lines 45-49",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "Citations are evaluated along three dimensions. (1) Link Works verifies URL accessibility, (2) Relevant Content measures topical alignment, and (3) Fact Check validates factual accuracy against source content.",

"quote_sha256": "d34096a9d9b1fbb43ef50ba503c3894607c008b563e1e1a5d1643cbc23c5d447",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-TERRA-03",

"source_id": "S1",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S1-A",

"span_occurrence_id": "V2-TERRA-03:S1:S1-A",

"span_root_id": "span:sha256:cd21c450c0d33df20a2b1a539d5725347db1b5bab7d1e9866c24790d845e5873",

"supports_declared_by_raw_answer": "metric definitions"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 1, lines 163-180",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "Model | Success | Link Works | Relevant | Fact Check",

"quote_sha256": "2dbc4a7649c02b8633c0a85f4073b013f1a5a9888905cb95135d7c2600082c38",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-TERRA-03",

"source_id": "S1",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S1-B",

"span_occurrence_id": "V2-TERRA-03:S1:S1-B",

"span_root_id": "span:sha256:35357cd8e4ff1531a84b4a5c8923db97c7b44fecd19d7666f326bbf7a1d5ec45",

"supports_declared_by_raw_answer": "the model-specific percentages and full-range comparison in R1"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Methods 3.4, lines 153-156",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "We evaluate 14 LLMs ... on 130 research queries drawn from DeepResearch Bench and BrowseComp.",

"quote_sha256": "75a8b058bb89da032df063c5b46c02062ba048b4fd46b396fe29d5020d5c60a8",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-TERRA-03",

"source_id": "S1",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S1-C",

"span_occurrence_id": "V2-TERRA-03:S1:S1-C",

"span_root_id": "span:sha256:d004f718815ff769195d087f1159e50cf5d00261b8ef85229625a65740d55b3d",

"supports_declared_by_raw_answer": "scope and shared-upstream-task warning"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Tables 2-3, lines 191-210",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "2 | 100.0% | 100.0% | 78.6% ... 150 | 99.2% | 99.2% | 16.7%",

"quote_sha256": "65918d19bdc3830f58ba9cb05e9cb9ae592d9409ffcaf50af9a5fefe7d843dc8",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-TERRA-03",

"source_id": "S1",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S1-D",

"span_occurrence_id": "V2-TERRA-03:S1:S1-D",

"span_root_id": "span:sha256:ec8d8cf5ad9b690ecd983e4d5c673d9624200351579e72868291abc15d0f2949",

"supports_declared_by_raw_answer": "GPT-5.4 depth ablation"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Evaluation 4.3, lines 211-215",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "Fact Check accuracy drops approximately 42% on average from minimal (2 calls) to maximal search depth.",

"quote_sha256": "d13421745ce8497cd47633527b808016a37fd65c43c761d448e325fc14e1ad73",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-TERRA-03",

"source_id": "S1",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S1-E",

"span_occurrence_id": "V2-TERRA-03:S1:S1-E",

"span_root_id": "span:sha256:fe6b18fa48b171b5bd7c74fda45f38ed1b087ee5035f6d35dcf8c701b3033ea8",

"supports_declared_by_raw_answer": "reported average depth comparison"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Limitations, lines 219-224",

"match_status": "ordered-fragment-match",

"matched_fragments": [

"the LLM-as-a-judge approach",

"may retain biases inherent to the judge model"

],

"quote": "the LLM-as-a-judge approach ... may retain biases inherent to the judge model",

"quote_sha256": "af0b066b292e6d887067c6cfa9802eb8b1984929b622d91cdca61d0e8a3bc31f",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-TERRA-03",

"source_id": "S1",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S1-F",

"span_occurrence_id": "V2-TERRA-03:S1:S1-F",

"span_root_id": "span:sha256:5cd9ba4ca72da10cf51c0beddc6ea2765c74721bdd526a67156f44754219f440",

"supports_declared_by_raw_answer": "judge and temporal-stability limitations"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Evaluation methodology, lines 144-149",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "Each unique Statement-URL pair undergoes a support evaluation.",

"quote_sha256": "1bde849b0f5df91e847e907dd763f27be01ffdc7eb0c7e61898c8a9731472c35",

"requested_url": "https://arxiv.org/html/2506.11763",

"run_id": "V2-TERRA-03",

"source_id": "S2",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S2-A",

"span_occurrence_id": "V2-TERRA-03:S2:S2-A",

"span_root_id": "span:sha256:015bb68c904bf225e84c82b1acffb3cc38a4a087d7b0f79d3b0102b685b2265e",

"supports_declared_by_raw_answer": "support metric methodology"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 1, lines 177-182",

"match_status": "ordered-fragment-match",

"matched_fragments": [

"Perplexity Deep Research |",

"| 90.24 | 31.26",

"Gemini-2.5-Pro Deep Research |",

"| 81.44 | 111.21",

"OpenAI Deep Research |",

"| 77.96 | 40.79"

],

"quote": "Perplexity Deep Research | ... | 90.24 | 31.26 ... Gemini-2.5-Pro Deep Research | ... | 81.44 | 111.21 ... OpenAI Deep Research | ... | 77.96 | 40.79",

"quote_sha256": "53d38f9cf0d0aedf28a77d7de32935372202fff78a0c02d98f8c269ef223906a",

"requested_url": "https://arxiv.org/html/2506.11763",

"run_id": "V2-TERRA-03",

"source_id": "S2",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S2-B",

"span_occurrence_id": "V2-TERRA-03:S2:S2-B",

"span_root_id": "span:sha256:0d4f357f2a5565bdaea01a1e537650958d71216868d1a2b0bc14c47a45b6d0ef",

"supports_declared_by_raw_answer": "R3 figures"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Appendix E, lines 404-423",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "Citation Accuracy ... [is] the proportion of 'support' statement-URL pairs ... averaging these per-task accuracies across all tasks.",

"quote_sha256": "74bf93e5475ed37377e1f4e679a9ef8d76ac73beba297cd3cfde4a3527052c76",

"requested_url": "https://arxiv.org/html/2506.11763",

"run_id": "V2-TERRA-03",

"source_id": "S2",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S2-C",

"span_occurrence_id": "V2-TERRA-03:S2:S2-C",

"span_root_id": "span:sha256:3ec266d89a7c774ebcee3d2b8b758ac56176c382481cd0d2f0bd893aaa142281",

"supports_declared_by_raw_answer": "quantitative basis"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Appendix A, lines 339-348",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "Each of our 100 tasks was developed by a verified PhD-level expert ... [but] inevitably constrains the dataset size.",

"quote_sha256": "2449d1c875bd9467747b4953e29dee8a011c23e0bc1535356161584206727f68",

"requested_url": "https://arxiv.org/html/2506.11763",

"run_id": "V2-TERRA-03",

"source_id": "S2",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S2-D",

"span_occurrence_id": "V2-TERRA-03:S2:S2-D",

"span_root_id": "span:sha256:de53fab3c5b2e262ba56cdd5a27eb002c2b4641755ccc38c734a5bd506a670ef",

"supports_declared_by_raw_answer": "scale and generalizability limitations"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Cited-statements pipeline, lines 121-124",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "we retrieve the full content of each cited webpage ... [and] perform consistency verification by comparing the statement with the retrieved content",

"quote_sha256": "5932bddfa8d2098b21caf7e4432af9726c4d258f8d52f8c0f9a77c077f76889d",

"requested_url": "https://arxiv.org/html/2508.15804",

"run_id": "V2-TERRA-03",

"source_id": "S3",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S3-A",

"span_occurrence_id": "V2-TERRA-03:S3:S3-A",

"span_root_id": "span:sha256:42a4184b2e116330dc263a3d8e9b77f4b88c9da1c325fc67c4375df0f818a514",

"supports_declared_by_raw_answer": "claim-source alignment method"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Settings and metrics, lines 132-139",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "For statement extraction, supporting source extraction, and semantic consistency verification, we adopt gpt-4o.",

"quote_sha256": "397d217cf5022ad78aa1072c2e82bcf3197e5a87cf4ef96ea81a78e1ad39b384",

"requested_url": "https://arxiv.org/html/2508.15804",

"run_id": "V2-TERRA-03",

"source_id": "S3",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S3-B",

"span_occurrence_id": "V2-TERRA-03:S3:S3-B",

"span_root_id": "span:sha256:ba6b8a8b7d9121ebb05d594176ceed3d1e9dbc637876783abfd0141ebbd0a822",

"supports_declared_by_raw_answer": "judge/tool scope"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 1, lines 140-150",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "OpenAI Deep Research | 0.385 | 0.033 | 9.89 | 78.87% | 88.2 ... Gemini Deep Research | 0.145 | 0.036 | 32.42 | 72.94% | 96.2",

"quote_sha256": "f62ae7f5e2d3de4804554378ea5745b0c2d100a1374ef638ec508efcb5bdb0b4",

"requested_url": "https://arxiv.org/html/2508.15804",

"run_id": "V2-TERRA-03",

"source_id": "S3",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S3-C",

"span_occurrence_id": "V2-TERRA-03:S3:S3-C",

"span_root_id": "span:sha256:7eefee71f11e6f5ec8c2e70272fd9189464650e68b904d9fa9c6d059bedc4195",

"supports_declared_by_raw_answer": "R4 quantitative figures"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Limitations, lines 259-261",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "The benchmark primarily draws from peer-reviewed survey papers on arXiv, most of which are concentrated in STEM fields.",

"quote_sha256": "4947edaf643bd0a234e0ca914d920a1a5b068d18b017dcfe6d6103c466d422ca",

"requested_url": "https://arxiv.org/html/2508.15804",

"run_id": "V2-TERRA-03",

"source_id": "S3",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S3-D",

"span_occurrence_id": "V2-TERRA-03:S3:S3-D",

"span_root_id": "span:sha256:1db12b68cfeb4c5f5a96fa9cd2e2bd46e8dfb4f096f73154481d29e22a4d5414",

"supports_declared_by_raw_answer": "domain limitation"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "Main text, paragraph beginning 'To address the gap' (search-indexed primary full text)",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "We imposed a 1500-word limit, APA-style references and conducted three independent runs.",

"quote_sha256": "1c94be957cfe750f445dc9143722905b5784b02f754a9b4071a3381e2bd493e9",

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"run_id": "V2-TERRA-03",

"source_id": "S4",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "S4-A",

"span_occurrence_id": "V2-TERRA-03:S4:S4-A",

"span_root_id": "span:sha256:b38a2af5982b4c85bc205d2f533a23ed3a8f40a49b651a5f00ec6932dcc67f7d",

"supports_declared_by_raw_answer": "study scope"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "Main text, following paragraph (search-indexed primary full text)",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "ChatGPT Deep Research-generated references were mostly identifiable by their metadata (95.7%): 69.6% were entirely correct",

"quote_sha256": "28141fd1b0222658eb93a338783358926de770223eea24f8c5cbcf632d67f955",

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"run_id": "V2-TERRA-03",

"source_id": "S4",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "S4-B",

"span_occurrence_id": "V2-TERRA-03:S4:S4-B",

"span_root_id": "span:sha256:b6a6ba48a2c4e352c8565b671a15a90400739f2aeefda8a65a31ad8a1ff9a4e4",

"supports_declared_by_raw_answer": "identifiability and correctness figures"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "Same paragraph (search-indexed primary full text)",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "fake authors or titles occurred frequently in references from Claude (95.8 +/- 7.2%), Gemini (47.6 +/- 7.8%) and Perplexity.AI (50.1 +/- 28.0%).",

"quote_sha256": "389559d433dae83a26d09f0ed85e1ca6761cee2d09850e974e215398d187d721",

"requested_url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13109748/",

"run_id": "V2-TERRA-03",

"source_id": "S4",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "S4-C",

"span_occurrence_id": "V2-TERRA-03:S4:S4-C",

"span_root_id": "span:sha256:f8ff0887370fedb6d6676d1eaf59f86461fe5083d20dfa5b28f8ba94175ed031",

"supports_declared_by_raw_answer": "fabrication-class figures"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-repo-main-469cce5",

"license_treatment": "metadata and quote-minimal README spans",

"locator": "README News, lines 176-188",

"match_status": "ordered-fragment-match",

"matched_fragments": [

"Official Evaluator Switched to GPT-5.5",

"Legacy code: the previous Gemini-2.5-Pro / Gemini-2.5-Flash evaluation code is preserved"

],

"quote": "Official Evaluator Switched to GPT-5.5 ... Legacy code: the previous Gemini-2.5-Pro / Gemini-2.5-Flash evaluation code is preserved",

"quote_sha256": "9c713be4fc872b46e66d812e033708b51b4e385c81aa45f7b1798459a677343d",

"requested_url": "https://github.com/Ayanami0730/deep_research_bench",

"run_id": "V2-TERRA-03",

"source_id": "S5",

"source_text_sha256": "afebe200cb45042b2783d6ad581770bbd8674027d4e7443e2eb6c26720c6cb0c",

"source_work_id": "work:deepresearch-bench-repository",

"span_id": "S5-A",

"span_occurrence_id": "V2-TERRA-03:S5:S5-A",

"span_root_id": "span:sha256:ef637b38f36195bea09205d532e8744daf3e257f93d97144f061024092bf67fe",

"supports_declared_by_raw_answer": "edition drift warning"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-repo-main-469cce5",

"license_treatment": "metadata and quote-minimal README spans",

"locator": "README FACT, lines 240-248",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "Support Verification: Uses web scraping and LLM judgment to verify whether cited sources actually support the claims",

"quote_sha256": "549018c6df261f27bd10133ad4e4f565e004f07710095259a7889cc79d149794",

"requested_url": "https://github.com/Ayanami0730/deep_research_bench",

"run_id": "V2-TERRA-03",

"source_id": "S5",

"source_text_sha256": "afebe200cb45042b2783d6ad581770bbd8674027d4e7443e2eb6c26720c6cb0c",

"source_work_id": "work:deepresearch-bench-repository",

"span_id": "S5-B",

"span_occurrence_id": "V2-TERRA-03:S5:S5-B",

"span_root_id": "span:sha256:92eb7cf19745e24190dda84c77d590941134d09f06805c0fb605207fd942ae11",

"supports_declared_by_raw_answer": "official description of FACT's intended function"

},

{

"citation_carrier_accessible": false,

"edition_id": "edition:keplinger-vor-2025",

"license_treatment": "quote-minimal attributed spans; no full-text redistribution",

"locator": "Figure 1 caption",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "Six inaccuracy types were identified within citation-bearing sentences ... Sentences may bear multiple types of inaccuracy, simultaneously.",

"quote_sha256": "1f9dd50794adad5378f0c97308d45a724c2f421f5cbfafd2d251d06bfb3b510e",

"requested_url": "https://pubmed.ncbi.nlm.nih.gov/40904191/",

"run_id": "V2-TERRA-03",

"source_id": "S6",

"source_text_sha256": "8fd80ac582690bc0733dec0ebd65e9894a05915d117384f3e8bb6d8b0840a7c1",

"source_work_id": "work:keplinger-dermatology-audit",

"span_id": "S6-A",

"span_occurrence_id": "V2-TERRA-03:S6:S6-A",

"span_root_id": "span:sha256:fdc35d731d751e7e0691158d8fed4a2934851897f9fef016b33376cde973827a",

"supports_declared_by_raw_answer": "failure-class and non-exclusive-category caution"

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 3.2, paragraphs 'Statement-URL Pair Extraction and Deduplication' and 'Support Judgment'",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"Each unique Statement-URL pair undergoes a support evaluation.\"",

"quote_sha256": "3cca518ed14d7451346a35623b20e7f9e350e4cb5cbcec3d7f622cc6aac509ce",

"requested_url": "https://arxiv.org/html/2506.11763",

"run_id": "V2-TERRA-04",

"source_id": "S1_deepresearchbench",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S1_method",

"span_occurrence_id": "V2-TERRA-04:S1_deepresearchbench:S1_method",

"span_root_id": "span:sha256:9fe42cb48702d8d96ebbbc0309692214ece6ec495c447c06e2be62f32ff00894",

"supports_declared_by_raw_answer": "The reported citation metric evaluates an extracted claim against retrieved cited-page content."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Table 1, Deep Research Agent rows",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"Perplexity Deep Research ... 90.24 ... 31.26\"",

"quote_sha256": "3997cfbb7dbefe6e98d2f77dd72ea2bda917c6838d9241e75139e7df312b149a",

"requested_url": "https://arxiv.org/html/2506.11763",

"run_id": "V2-TERRA-04",

"source_id": "S1_deepresearchbench",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S1_table1",

"span_occurrence_id": "V2-TERRA-04:S1_deepresearchbench:S1_table1",

"span_root_id": "span:sha256:539d8123b83aa1250f16cb7a8a4a66ece5d42099c0ae7c7d1fd877f1fbafb3bd",

"supports_declared_by_raw_answer": "Perplexity's FACT citation accuracy and effective-citation result; the adjacent rows provide Gemini, OpenAI, and Grok values."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Appendix E.1, equations (4)-(5)",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"C. Acc. assesses the ... proportion of 'support' statement-URL pairs\"",

"quote_sha256": "d7619da3b090b3608e35b23fb944490f090732ac0f42a4bdfc73a90f43094bdb",

"requested_url": "https://arxiv.org/html/2506.11763",

"run_id": "V2-TERRA-04",

"source_id": "S1_deepresearchbench",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S1_metric",

"span_occurrence_id": "V2-TERRA-04:S1_deepresearchbench:S1_metric",

"span_root_id": "span:sha256:a13bd114b7a394950897d477dccb999910794b70914bbc3a45fa6c4667181992",

"supports_declared_by_raw_answer": "The denominator and macro-averaging definition."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:drbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Appendix D, Table 5",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"OpenAI Deep Research | April 1 \u2013 May 8\"",

"quote_sha256": "86d5d20702fbe736e87099be370fadf16b4acea42bea16ba28ebb473ba2ae670",

"requested_url": "https://arxiv.org/html/2506.11763",

"run_id": "V2-TERRA-04",

"source_id": "S1_deepresearchbench",

"source_text_sha256": "c8391dc3b757d97aa6d781636819180d34cb7343d290896d54db43f6f467da89",

"source_work_id": "work:deepresearch-bench-paper",

"span_id": "S1_time",

"span_occurrence_id": "V2-TERRA-04:S1_deepresearchbench:S1_time",

"span_root_id": "span:sha256:76e9bd59b31c7ebd2ed7a2186e59753a79d3c730e0d7b4fbf0146fb39f4e9c5f",

"supports_declared_by_raw_answer": "Collection-time scope for the evaluated commercial outputs."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:researcherbench-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Section 4.2.1 'Citation Support Verification' and Section 4.2 'Score Computation', equations (2)-(3)",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"whether the extracted content supports the corresponding claim\"",

"quote_sha256": "12525ade29e741070aec9df092b69ee828e7f4333a4c41c331784cfab48bef56",

"requested_url": "https://arxiv.org/html/2507.16280",

"run_id": "V2-TERRA-04",

"source_id": "S2_researcherbench",

"source_text_sha256": "1aa77342c2571bb564fe1b3e9a2ba2333b4ef0d637c4a3ef622540741927b0ec",

"source_work_id": "work:researcherbench",

"span_id": "S2_method",

"span_occurrence_id": "V2-TERRA-04:S2_researcherbench:S2_method",

"span_root_id": "span:sha256:f96ad5daa81153a3689f52f5777fedfd70e8a87d44e7a1c9e2c22b5789a3707d",

"supports_declared_by_raw_answer": "The study's faithfulness numerator is direct claim-source support."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:researcherbench-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Table 2, Deep Research System rows",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"OpenAI Deep Research | 0.7032 | 0.84 | 0.34\"",

"quote_sha256": "5c9ca63ca1d8757a3e9e598d441fbca1675657a026bb4f97424112eded86594d",

"requested_url": "https://arxiv.org/html/2507.16280",

"run_id": "V2-TERRA-04",

"source_id": "S2_researcherbench",

"source_text_sha256": "1aa77342c2571bb564fe1b3e9a2ba2333b4ef0d637c4a3ef622540741927b0ec",

"source_work_id": "work:researcherbench",

"span_id": "S2_table2",

"span_occurrence_id": "V2-TERRA-04:S2_researcherbench:S2_table2",

"span_root_id": "span:sha256:5dbd4e3e48480f5b3c54ce606082c35bfd55bf40cd0be847cc03a8fd4d9993f6",

"supports_declared_by_raw_answer": "OpenAI's coverage, faithfulness, and groundedness values; adjacent rows report Gemini, Grok, and Perplexity."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:researcherbench-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Abstract and Section 5.1 'Evaluation Configuration'",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"65 research questions ... across 35 different AI subjects\"",

"quote_sha256": "02c896e17544a2a9ab762d11facbfda91495f997bcf02560161c971a40ae78fa",

"requested_url": "https://arxiv.org/html/2507.16280",

"run_id": "V2-TERRA-04",

"source_id": "S2_researcherbench",

"source_text_sha256": "1aa77342c2571bb564fe1b3e9a2ba2333b4ef0d637c4a3ef622540741927b0ec",

"source_work_id": "work:researcherbench",

"span_id": "S2_scope",

"span_occurrence_id": "V2-TERRA-04:S2_researcherbench:S2_scope",

"span_root_id": "span:sha256:0d70187dbf9e3044ae9efd387faf34b29b9ab556aae6b0ea4c06e05964bfa2c7",

"supports_declared_by_raw_answer": "Task-domain and population scope."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:researcherbench-arxiv-v1",

"license_treatment": "metadata, digest, and quote-minimal attributed spans only",

"locator": "Section 5.1 and Appendix E, Table 4",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"All evaluations were conducted between March and April in 2025\"",

"quote_sha256": "1eb2dcb713304fb76409f4ea450d6b06ba05e95e02110aed33dad456ab3d8ce8",

"requested_url": "https://arxiv.org/html/2507.16280",

"run_id": "V2-TERRA-04",

"source_id": "S2_researcherbench",

"source_text_sha256": "1aa77342c2571bb564fe1b3e9a2ba2333b4ef0d637c4a3ef622540741927b0ec",

"source_work_id": "work:researcherbench",

"span_id": "S2_time",

"span_occurrence_id": "V2-TERRA-04:S2_researcherbench:S2_time",

"span_root_id": "span:sha256:ea9cb2e1b0078c0857b789d61d0a0d26801cf6689186470588a87b98b4639796",

"supports_declared_by_raw_answer": "Evaluation period and version-boundary caution."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 2.2 'Cited statements'",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"retrieve the full content of each cited webpage\"",

"quote_sha256": "153b2f37a8662ac0dbbd7fe51bcbb291afd5d227c4da3f4c6463274c63a11fb2",

"requested_url": "https://arxiv.org/html/2508.15804",

"run_id": "V2-TERRA-04",

"source_id": "S3_reportbench",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S3_method",

"span_occurrence_id": "V2-TERRA-04:S3_reportbench:S3_method",

"span_root_id": "span:sha256:9c7a4c186cc85f185aa293a7c5a46c08f9dbbc1a9e63c449dc737447e42f8ee4",

"supports_declared_by_raw_answer": "The report measures cited-statement support against retrieved cited content."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 3.2, Table 1",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"OpenAI Deep Research ... 78.87% ... 88.2\"",

"quote_sha256": "2670aff9a658c1e206579253b86b7ce0d543c6bd1bede985279faa0a3df18b27",

"requested_url": "https://arxiv.org/html/2508.15804",

"run_id": "V2-TERRA-04",

"source_id": "S3_reportbench",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S3_table1",

"span_occurrence_id": "V2-TERRA-04:S3_reportbench:S3_table1",

"span_root_id": "span:sha256:9a2cf22761f2ebb97eb31dfa02a32775248c3a7396b1e410565c313d30585cf3",

"supports_declared_by_raw_answer": "OpenAI match rate and average cited-statement count; Gemini's adjacent row gives 72.94% and 96.2."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 3.1 'Setttings'",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"during the period from July 14 to July 25\"",

"quote_sha256": "b55cc3d05dbba5d160b75515ce407128fe2e2cf8c2d679517e79309d6ed6ab40",

"requested_url": "https://arxiv.org/html/2508.15804",

"run_id": "V2-TERRA-04",

"source_id": "S3_reportbench",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S3_time",

"span_occurrence_id": "V2-TERRA-04:S3_reportbench:S3_time",

"span_root_id": "span:sha256:ddf4320a6d302c13d28b6225eb7e79ac15583349e1b449d4bb967eb89a257a3d",

"supports_declared_by_raw_answer": "Data-collection dates and WebUI setup."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:reportbench-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Appendix A.1 'Limitations'",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"most of which are concentrated in STEM fields\"",

"quote_sha256": "ee6c65565e833960bd4ef6552fe3153a1108a743989a21b261b571cf6f066aa4",

"requested_url": "https://arxiv.org/html/2508.15804",

"run_id": "V2-TERRA-04",

"source_id": "S3_reportbench",

"source_text_sha256": "014f247563e015dfe7cd6ea4dee6c410f57c74788a6fb5bd589f65df078e8c95",

"source_work_id": "work:reportbench",

"span_id": "S3_limitations",

"span_occurrence_id": "V2-TERRA-04:S3_reportbench:S3_limitations",

"span_root_id": "span:sha256:084f11e57cf779850ea87aaf30ba1277438ff9efcb533fd2ebb09dba3665be8d",

"supports_declared_by_raw_answer": "Dataset-domain limitation."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 4.1, Figure 1/Table 2",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"openai-deepresearch | OpenAI | 4,121 | 10.1 ... | 3.5 ... | 6.6\"",

"quote_sha256": "fa5bbff787d875e995bd714b8b2f60a9a3cc9ece9e99d6af5f47861aa54e3a97",

"requested_url": "https://arxiv.org/html/2604.03173",

"run_id": "V2-TERRA-04",

"source_id": "S4_urlhealth",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "S4_table2",

"span_occurrence_id": "V2-TERRA-04:S4_urlhealth:S4_table2",

"span_root_id": "span:sha256:131799471acf2f83cc3a17322a1c5cda6b5b46020bd894be54a1708db62eb14b",

"supports_declared_by_raw_answer": "OpenAI Deep Research URL denominator and non-resolving/hallucinated/stale results; the Gemini row is adjacent."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 4.2, paragraph beginning 'Deep research agents cite far more URLs per query'",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"16.2% ... versus 6.8%\"",

"quote_sha256": "f667344d15b6ab97def48b5c4b60af01b003048a652166d1c27045c53bcc1c74",

"requested_url": "https://arxiv.org/html/2604.03173",

"run_id": "V2-TERRA-04",

"source_id": "S4_urlhealth",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "S4_comparison",

"span_occurrence_id": "V2-TERRA-04:S4_urlhealth:S4_comparison",

"span_root_id": "span:sha256:dc108a5e00fb6ce5306fcde8fd0c36c85352dac361662fab3ad1eea89d66c048",

"supports_declared_by_raw_answer": "Pooled deep-research versus search-augmented non-resolution comparison."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 3.3 'URL extraction and classification'",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"URLs returning 4xx or 5xx status codes, connection errors, or timeouts are classified as non-resolving\"",

"quote_sha256": "b240bd208cb4c322f3c9c242d01e13da53b3fd2a64e41b8b0f26653032e0ee54",

"requested_url": "https://arxiv.org/html/2604.03173",

"run_id": "V2-TERRA-04",

"source_id": "S4_urlhealth",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "S4_method",

"span_occurrence_id": "V2-TERRA-04:S4_urlhealth:S4_method",

"span_root_id": "span:sha256:bb62928b5e0cb4e373b2f6bfeb579d8564d05da5588f8b150a89ea4b9b2ed0db",

"supports_declared_by_raw_answer": "Operational definition of URL resolution and Wayback-based stale/hallucinated split."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:url-health-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 2 definition discussion and Section 3.3",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"our non-resolving rates are lower bounds\"",

"quote_sha256": "19f5203d42cda0ad9b817dc8b5e7abf1b449fc7fa9792a8cdcbf6ea8782349b1",

"requested_url": "https://arxiv.org/html/2604.03173",

"run_id": "V2-TERRA-04",

"source_id": "S4_urlhealth",

"source_text_sha256": "d00cf613da17939694913654e404c5d61e78c62779fdd9d154655f5986fe643f",

"source_work_id": "work:url-health",

"span_id": "S4_limitations",

"span_occurrence_id": "V2-TERRA-04:S4_urlhealth:S4_limitations",

"span_root_id": "span:sha256:59d0da64f738e1d8745f2bff3663eb89e273f93fe501afe37e2494a8e0ff31a1",

"supports_declared_by_raw_answer": "403/bot-blocking and classification limitation."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 3.3.1-3.3.3",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"Fact Check verifies whether specific factual claims are accurately supported by the source content.\"",

"quote_sha256": "a4386d56325583893d806fdd15ccd475d5096f0b445b1b9f52aeba253c910987",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-TERRA-04",

"source_id": "S5_cited_not_verified",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S5_method",

"span_occurrence_id": "V2-TERRA-04:S5_cited_not_verified:S5_method",

"span_root_id": "span:sha256:732c03e69c6e142a492275cee5d1e2d474c75422c9b84aaaf273100e937338ed",

"supports_declared_by_raw_answer": "Direct support criterion and distinction from URL accessibility/relevance."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 3.4 'Experimental setup'",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"14 LLMs ... on 130 research queries drawn from DeepResearch Bench and BrowseComp\"",

"quote_sha256": "d8bd0561648c1f5ac4ca416fa90891f186ae248737edc9d973703ea20b1231d4",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-TERRA-04",

"source_id": "S5_cited_not_verified",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S5_scope",

"span_occurrence_id": "V2-TERRA-04:S5_cited_not_verified:S5_scope",

"span_root_id": "span:sha256:cebc739ec2bf421e0853467bfeccb3f42bda1cf05472076a1c1bc6d9fb1edfb9",

"supports_declared_by_raw_answer": "Model and dataset scope."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 4.1, Table 1",

"match_status": "unresolved",

"matched_fragments": [],

"quote": "\"Claude Opus 4.5 | 90.0% | 98.7% | 95.7% | 76.8%\"",

"quote_sha256": "42b7ccd7579ae20fa667410ef47609278f518af058f6feb1b15409b7fb253d01",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-TERRA-04",

"source_id": "S5_cited_not_verified",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S5_table1",

"span_occurrence_id": "V2-TERRA-04:S5_cited_not_verified:S5_table1",

"span_root_id": "span:sha256:caaa3e2fbb4cef64558f2c8cc30388281e1505c9353fb775f960a5f18c816111",

"supports_declared_by_raw_answer": "Success, accessibility, relevance, and Fact Check rates; adjacent rows report the remaining models."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 4.3, Tables 2-3 and following paragraph",

"match_status": "exact-normalized-match",

"matched_fragments": [],

"quote": "\"Fact Check accuracy drops approximately 42% on average\"",

"quote_sha256": "4ff735800b0f2aca92c4c6865a7bcf5746c1b7711a6105d5403bdffb68adce63",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-TERRA-04",

"source_id": "S5_cited_not_verified",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S5_depth",

"span_occurrence_id": "V2-TERRA-04:S5_cited_not_verified:S5_depth",

"span_root_id": "span:sha256:d638c8b21f0e4a0f840880b6ffe0217ce1623758cc09fdccc32cacf2db3dd958",

"supports_declared_by_raw_answer": "Search-depth ablation summary."

},

{

"citation_carrier_accessible": true,

"edition_id": "edition:cnv-arxiv-v1",

"license_treatment": "quote-minimal attributed spans",

"locator": "Section 5 'Limitations'",

"match_status": "ordered-fragment-match",

"matched_fragments": [

"LLM-as-a-judge",

"may retain biases inherent to the judge model"

],

"quote": "\"LLM-as-a-judge ... may retain biases inherent to the judge model\"",

"quote_sha256": "a14ffeb203983bb9ab414d79300766c78a29ade822b77d94bdd51c6788cfe450",

"requested_url": "https://arxiv.org/html/2605.06635",

"run_id": "V2-TERRA-04",

"source_id": "S5_cited_not_verified",

"source_text_sha256": "1dd04031e118646bdebec5e35d289cc625e758ff165a63ba3cbf0680ad12bb10",

"source_work_id": "work:cited-not-verified",

"span_id": "S5_limitations",

"span_occurrence_id": "V2-TERRA-04:S5_cited_not_verified:S5_limitations",

"span_root_id": "span:sha256:c7dbd1d9e09980703ecbee8161594e4b426c1af63117b28662a7e98686107621",

"supports_declared_by_raw_answer": "Judge-based support-evaluation limitation."

}

],

"status": "research-only-author-review-complete-independent-review-required",

"task_id": "EM-0026",

"works": [

{

"data_root": "data:citationverbench-1248-human-reviewed-decisions",

"derivation_root": "derivation:citationverbench-reported-calibration",

"identifiers": [

"arxiv:2607.08700"

],

"method_root": "method:citationverbench-rubric-judge-comparison",

"title": "Do You Need a Frontier Model as a Citation Verifier? Benchmarking Rubric LLMs for Deep-Research Source Attribution",

"work_id": "work:citation-verifier-benchmark"

},

{

"data_root": "data:cnv-130-drbench-and-browsecomp-queries",

"derivation_root": "derivation:cnv-main-and-depth-ablation",

"identifiers": [

"arxiv:2605.06635"

],

"method_root": "method:cnv-ast-access-relevance-fact-check",

"title": "Cited but Not Verified: Parsing and Evaluating Source Attribution in LLM Deep Research Agents",

"work_id": "work:cited-not-verified"

},

{

"data_root": "data:deepresearch-bench-100-tasks-and-system-outputs",

"derivation_root": "derivation:deepresearch-bench-edition-specific-tables",

"identifiers": [

"arxiv:2506.11763",

"openreview:hQ0K2Hhq7H"

],

"method_root": "method:fact-statement-url-support-evaluator",

"title": "DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents",

"work_id": "work:deepresearch-bench-paper"

},

{

"data_root": "data:deepresearch-bench-100-tasks-and-system-outputs",

"derivation_root": "derivation:repository-main-469cce54ea7f6a63c163d3d9fec879cf289ec484",

"identifiers": [

"github:Ayanami0730/deep_research_bench"

],

"method_root": "method:fact-reference-implementation",

"title": "Ayanami0730/deep_research_bench",

"work_id": "work:deepresearch-bench-repository"

},

{

"data_root": "data:deeptrace-303-questions-2727-outputs",

"derivation_root": "derivation:deeptrace-tables-and-prose",

"identifiers": [

"arxiv:2509.04499",

"iclr:ad08767706825033b99122332293033d"

],

"method_root": "method:deeptrace-statement-source-matrix",

"title": "DeepTRACE: Auditing Deep Research AI Systems for Tracking Reliability Across Citations and Evidence",

"work_id": "work:deeptrace"

},

{

"data_root": "data:keplinger-three-run-five-system-audit",

"derivation_root": "derivation:keplinger-published-tables-and-figure",

"identifiers": [

"doi:10.1111/jdv.70035",

"pmcid:PMC13109748",

"pmid:40904191"

],

"method_root": "method:human-reference-and-sentence-concordance-audit",

"title": "Assessment of Deep Research for dermatology literature reviews: Deep concern over the hype",

"work_id": "work:keplinger-dermatology-audit"

},

{

"data_root": "data:keplinger-three-run-five-system-audit",

"derivation_root": "derivation:keplinger-supplement-files",

"identifiers": [

"doi:10.17632/3s73z9zf3c"

],

"method_root": "method:keplinger-supplementary-prompts-and-results",

"title": "Supplementary materials of the article: Assessment of Deep Research for Dermatology Literature Reviews: Deep Concern Over the Hype",

"work_id": "work:keplinger-supplement"

},

{

"data_root": "data:liveresearchbench-wide-info-and-market-analysis",

"derivation_root": "derivation:liveresearchbench-v5-table-7",

"identifiers": [

"arxiv:2510.14240"

],

"method_root": "method:liveresearchbench-e1-e2-e3-rubric-tree",

"title": "LiveResearchBench: A Live Benchmark for User-Centric Deep Research in the Wild",

"work_id": "work:liveresearchbench"

},

{

"data_root": "data:reportbench-100-survey-tasks",

"derivation_root": "derivation:reportbench-table-1",

"identifiers": [

"arxiv:2508.15804"

],

"method_root": "method:reportbench-statement-source-semantic-match",

"title": "ReportBench: Evaluating Deep Research Agents via Academic Survey Tasks",

"work_id": "work:reportbench"

},

{

"data_root": "data:researcherbench-65-frontier-ai-questions",

"derivation_root": "derivation:researcherbench-table-2",

"identifiers": [

"arxiv:2507.16280"

],

"method_root": "method:researcherbench-faithfulness-groundedness",

"title": "ResearcherBench: Evaluating Deep AI Research Systems on the Frontiers of Scientific Inquiry",

"work_id": "work:researcherbench"

},

{

"data_root": "data:url-health-drbench-and-expertqa-urls",

"derivation_root": "derivation:url-health-observational-and-correction-results",

"identifiers": [

"arxiv:2604.03173"

],

"method_root": "method:http-resolution-wayback-classification",

"title": "Detecting and Correcting Reference Hallucinations in Commercial LLMs and Deep Research Agents",

"work_id": "work:url-health"

}

]

}

Build receipt

Reproduce this projection

Reproducible projection
Catalog
em:catalog:sha256:9bfc972213cba2cde167386103dc2c011ee74639fb7f0794c54120fbbdef1a5d
Frontier
em:frontier:sha256:f33be3eae4c75232d56750ef9a1aa79d96274ece3417d65a75c1391bf61a81bf
Accepted commit
f92846570180dfa4511263f8ba98ecd18f7772c9
Epistemic policy
commons-balanced-v0.1
Disclosure policy
public-noninterference-v0.1
Compiler
epistemedia/0.2.0