{"as_of":"2026-08-07T08:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:39f9fadf2584b19e0bb10ce4088a21aa20d5c071fbf407a9f5b7cf6b8f6b608e","coverage":[{"denominator":77,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":77,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T19:46:45.179215Z","state":"measured"},{"denominator":82,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":82,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T13:31:24.303475Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-03T10:27:56.511231Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"cited_work":{"arxiv_id":"2603.01152","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.01152","snapshot_observed_at":"2026-07-03T10:27:56.511231Z","title":"Deepresearch-9k: A challenging benchmark dataset of deep-research agent","venue":"cs.AI","work_id":"62b4fe1a-19ad-47fa-9046-89f23257ecdb","year":2026},"citing_paper":{"arxiv_id":"2604.17265","last_updated":"2026-05-12T13:20:03Z","snapshot_observed_at":"2026-08-02T13:37:40.268419Z","submitted_at":"2026-04-19T05:35:06Z","title":"MemSearch-o1: Empowering Large Language Models with Reasoning-Aligned Memory Growth in Agentic Search","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-10T06:15:11.432788Z"},"links":{"cited_paper":"/paper/2603.01152","citing_paper":"/paper/2604.17265"},"observation_digest":"sha256:f384a5e7a6c7bc4d6b32bde59dc30b57cbb53109897e44ea3b6602ba713c3899","observation_id":"7d98f528-00f5-4c36-8abd-e32f7a39e76c","resolution":{"observed_at":"2026-06-23T03:12:46.256547Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"cited_work":{"arxiv_id":"2603.01152","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.01152","snapshot_observed_at":"2026-07-03T10:27:56.511231Z","title":"Deepresearch-9k: A challenging benchmark dataset of deep-research agent","venue":"cs.AI","work_id":"62b4fe1a-19ad-47fa-9046-89f23257ecdb","year":2026},"citing_paper":{"arxiv_id":"2605.21965","last_updated":"2026-05-21T03:55:47Z","snapshot_observed_at":"2026-07-06T23:32:20.708664Z","submitted_at":"2026-05-21T03:55:47Z","title":"SpecHop: Continuous Speculation for Accelerating Multi-Hop Retrieval Agents","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-22T06:50:06.933671Z"},"links":{"cited_paper":"/paper/2603.01152","citing_paper":"/paper/2605.21965"},"observation_digest":"sha256:dcdde74bdc65bca6683b37a7d1d41048f1e0b2621ccef2b446baf30da289bd34","observation_id":"3e54a3a1-ad33-41c7-8a04-6163fae6ff69","resolution":{"observed_at":"2026-06-23T03:12:46.256547Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"cited_work":{"arxiv_id":"2603.01152","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.01152","snapshot_observed_at":"2026-07-03T10:27:56.511231Z","title":"Deepresearch-9k: A challenging benchmark dataset of deep-research agent","venue":"cs.AI","work_id":"62b4fe1a-19ad-47fa-9046-89f23257ecdb","year":2026},"citing_paper":{"arxiv_id":"2606.12087","last_updated":"2026-06-10T13:49:11Z","snapshot_observed_at":"2026-08-02T15:03:50.335992Z","submitted_at":"2026-06-10T13:49:11Z","title":"FORT-Searcher: Synthesizing Shortcut-Resistant Search Tasks for Training Deep Search Agents","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-06-27T10:01:45.332920Z"},"links":{"cited_paper":"/paper/2603.01152","citing_paper":"/paper/2606.12087"},"observation_digest":"sha256:17fa5d06b429e6c219e98d494a39c9b89c57331528eaae1bf6b29cfd86a0ef8d","observation_id":"99936593-f8fe-4c87-aed6-23bcfaebd5e5","resolution":{"observed_at":"2026-07-03T10:27:56.512551Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.01152","snapshot_observed_at":"2026-08-01T15:02:37.058668Z","title":"Deepresearch-9k: A challenging benchmark dataset of deep-research agent.arXiv preprint arXiv:2603.01152, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.28098","last_updated":"2026-07-29T05:08:21Z","snapshot_observed_at":"2026-08-03T00:03:31.143635Z","submitted_at":"2026-07-29T05:08:21Z","title":"SciDataSailor: Deep Scientific Data Exploring","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T15:02:37.058668Z"},"links":{"cited_paper":"/paper/2603.01152","citing_paper":"/paper/2607.28098"},"observation_digest":"sha256:3b39dc831d5b13d32206477c7e9b19aa2b13bc0c2d63391544f489166f37b5fe","observation_id":"7316b579-fadf-432c-993a-8c296e06f9a3","resolution":{"observed_at":"2026-08-01T15:02:37.058668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.01152","snapshot_observed_at":"2026-08-04T13:31:24.303475Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.02163","last_updated":"2026-08-03T12:43:00Z","snapshot_observed_at":"2026-08-06T23:36:58.187063Z","submitted_at":"2026-08-03T12:43:00Z","title":"From Simple QA to Deep Research: A Verifiable Benchmark Constructed through Iterative Task Evolution","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-04T13:31:24.303475Z"},"links":{"cited_paper":"/paper/2603.01152","citing_paper":"/paper/2608.02163"},"observation_digest":"sha256:f400d44a9223b576da02d096500a5fe2664fa8dc51065d0cc7a0bdb1f6ddbc33","observation_id":"e84c7fd4-68bc-482a-9909-2e0e918727ec","resolution":{"observed_at":"2026-08-04T13:31:24.303475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2603.01152/citation-record","integrity":"/paper/2603.01152/integrity","json":"/paper/2603.01152/citation-record.json","paper":"/paper/2603.01152"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:36.056202Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:36.056202Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:e9bc4fb773619e793731a0cf54c78d4307bfd287f4516714935ab99a0e4a960f","observation_id":"47e9dc36-23cf-4350-a2ba-eb591cc4ea86","resolution":{"observed_at":"2026-08-02T19:46:36.056202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:36.264784Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:36.264784Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:f661c7886207ab437cacf5f7939e1dbad14f55681e24afc1536b39f8f4585d43","observation_id":"80a19db1-52a9-40c4-bcc3-38ddb0f1f0c0","resolution":{"observed_at":"2026-08-02T19:46:36.264784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:36.390170Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:36.390170Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:a9bb721750982f6fb84927eda39c20cca3f031b64c908ec67ba0dff6c6d5b5b4","observation_id":"3297945f-d0b2-47c1-9c77-8e258456db4f","resolution":{"observed_at":"2026-08-02T19:46:36.390170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:36.547474Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:36.547474Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:b8fa56ff90d3508aeb975b510cb6dd697d00883bf8b3fb0b35c548414d2e1cb4","observation_id":"4f11f2fc-37c3-446f-b122-27214ce49a1a","resolution":{"observed_at":"2026-08-02T19:46:36.547474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:36.681856Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:36.681856Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:c716721daeecccb13fe369cbef2ad3514483af00589009053d5502f0f5644178","observation_id":"813ba9b5-bccb-4c27-bc6f-8ead17453430","resolution":{"observed_at":"2026-08-02T19:46:36.681856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:36.809506Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:36.809506Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:3ecf3beedd5dd38b66d3f8b6b6c1a8d399e26a587085a82e3107f783b23db136","observation_id":"3dc44489-6808-4722-8d84-a011fdcdad5a","resolution":{"observed_at":"2026-08-02T19:46:36.809506Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.06600","last_updated":"2025-08-08T17:55:11Z","snapshot_observed_at":"2026-08-07T04:31:29.392346Z","submitted_at":"2025-08-08T17:55:11Z","title":"BrowseComp-Plus: A More Fair and Transparent Evaluation Benchmark of Deep-Research Agent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.06600","snapshot_observed_at":"2026-08-02T19:46:36.985526Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:36.985526Z"},"links":{"cited_paper":"/paper/2508.06600","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:8e7a2032f0fff20b5741f5d55d9ced9be895aa3dd232578684a63cba1f7fa347","observation_id":"4960c364-c247-4542-b649-0a74dd02ab3a","resolution":{"observed_at":"2026-08-02T19:46:36.985526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:37.150401Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:37.150401Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:07a05b7a6085f5f99d931c8494ac7cfb18665f27db593cc63eb5e538637ac135","observation_id":"a3c1c305-3f1e-495c-ae83-50f4220ef2dd","resolution":{"observed_at":"2026-08-02T19:46:37.150401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-08-02T19:46:37.307887Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:37.307887Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:c3ccd897be845a115ed29a4787a455782a7743967918cf3fd50da21d22c0d202","observation_id":"9f07014f-9e79-4513-a516-698b07620042","resolution":{"observed_at":"2026-08-02T19:46:37.307887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11763","last_updated":"2025-06-13T13:17:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-13T13:17:32Z","title":"DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11763","snapshot_observed_at":"2026-08-02T19:46:37.475191Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:37.475191Z"},"links":{"cited_paper":"/paper/2506.11763","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:ec39bf7a4d46cb5f6a03bebcc1424ab62ac4e7b8689d751e32e68c4076b520b9","observation_id":"10b426c9-630f-461d-8675-2bebe77d74cb","resolution":{"observed_at":"2026-08-02T19:46:37.475191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:37.610798Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:37.610798Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:a7bd64cd1aabfa7da31f90f2b6fe7082a158e7a7b48881c076b44dd1abe7f36b","observation_id":"33dc1d2e-cf51-4084-90ec-76791cab9f59","resolution":{"observed_at":"2026-08-02T19:46:37.610798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.10504","last_updated":"2026-07-12T05:28:34Z","snapshot_observed_at":"2026-08-03T10:21:03.233752Z","submitted_at":"2026-01-15T15:28:21Z","title":"DR-Arena: an Automated Evaluation Framework for Deep Research Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.10504","snapshot_observed_at":"2026-08-02T19:46:37.784718Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:37.784718Z"},"links":{"cited_paper":"/paper/2601.10504","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:291c0a945a1471c709deab779f57b20c9e0be9d7d47901de854afe18540fed0c","observation_id":"8bf90c39-ffc7-4b80-8814-5b844ab97720","resolution":{"observed_at":"2026-08-02T19:46:37.784718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:37.961831Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:37.961831Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:712492803b4a91e830eb7fb822ddf177f72304e6f19695ab1e20644726c7f6ee","observation_id":"1aea910a-f0b2-4fbb-ad40-61aee70d5d6b","resolution":{"observed_at":"2026-08-02T19:46:37.961831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:38.074076Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:38.074076Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:957576c0e2af4f6e9ecc432448dac396529be3ee9181ca15a31c6592b8be8935","observation_id":"a4da654d-ad73-4c04-a5d7-f610268a9f8d","resolution":{"observed_at":"2026-08-02T19:46:38.074076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:38.240465Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:38.240465Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:1dd025d3e679c9c570633ad1f03333b2aa316e616bb56c44fc2d1a6d5dd17416","observation_id":"7a82a6fa-ad98-47a2-8b64-9418908a5cc8","resolution":{"observed_at":"2026-08-02T19:46:38.240465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:38.376216Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:38.376216Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:7f711bc86a4252c90578d5a667528101c19a8c132737c28f0eb50b60400d829c","observation_id":"cbce755f-07ca-49b0-a44f-930db2fba3ab","resolution":{"observed_at":"2026-08-02T19:46:38.376216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-02T19:46:38.461831Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:38.461831Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:c895e9ee926a710471e9542c45bcc6173860d63c25b8aadaa9de2fdbba8c6c69","observation_id":"e126681a-8d32-4fc2-ab2a-c11c6f4e1f00","resolution":{"observed_at":"2026-08-02T19:46:38.461831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2011.01060","last_updated":"2020-11-12T07:47:48Z","snapshot_observed_at":"2026-07-06T10:10:56.466018Z","submitted_at":"2020-11-02T15:42:40Z","title":"Constructing A Multi-hop QA Dataset for Comprehensive Evaluation of Reasoning Steps","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2011.01060","snapshot_observed_at":"2026-08-02T19:46:38.547858Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:38.547858Z"},"links":{"cited_paper":"/paper/2011.01060","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:621224e2f31ec87117482eee382d81c56e015d71740e24a3984efde115a8722c","observation_id":"ccecdd0d-125a-4631-a6ce-5cd4740c5371","resolution":{"observed_at":"2026-08-02T19:46:38.547858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:38.648298Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:38.648298Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:57c9e666785c03915e5e2a9d64d9841054d72d8c7a6b5bd2f2c5578ee27cd38c","observation_id":"c69d740a-8abf-46b0-b39c-c860bba24388","resolution":{"observed_at":"2026-08-02T19:46:38.648298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09516","last_updated":"2025-08-05T19:08:38Z","snapshot_observed_at":"2026-07-06T20:51:28.022519Z","submitted_at":"2025-03-12T16:26:39Z","title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09516","snapshot_observed_at":"2026-08-02T19:46:38.734248Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:38.734248Z"},"links":{"cited_paper":"/paper/2503.09516","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:86bd89bdd64357bc2c6be133bbc3646743d42864dabf2431cdae69b67663c19d","observation_id":"d59c7722-b03b-48a0-b7af-2245a51ec768","resolution":{"observed_at":"2026-08-02T19:46:38.734248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:38.890415Z","title":null,"venue":null,"work_id":null,"year":1996},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:38.890415Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:016b417871ec1d5b90461627c180410ccc72d097ccf8a6b359f6459d79691577","observation_id":"c952bda9-4990-4e36-9a31-7513ea897b80","resolution":{"observed_at":"2026-08-02T19:46:38.890415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:39.014421Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:39.014421Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:ede791648bca9987ce17e07dd4fc38e9a47ab4e6220d7294d506b829389e36dd","observation_id":"6fc26cbb-da25-49d5-98b4-cf9163f6958f","resolution":{"observed_at":"2026-08-02T19:46:39.014421Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:39.130399Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:39.130399Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:9db61aa0b63871065be5814839a7477ff2154052bfb9e60b3500320168142b74","observation_id":"96ddb04e-0bf2-47b5-95ac-3e06ca604db2","resolution":{"observed_at":"2026-08-02T19:46:39.130399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:39.256277Z","title":"Gonzalez, Hao Zhang, and Ion Stoica","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:39.256277Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:eeaad18830f8074328305ba77a41446a3a282f5ec13aeb0f150b104f235b8d22","observation_id":"aa529ef4-2660-4717-9c6f-1c6835f7834f","resolution":{"observed_at":"2026-08-02T19:46:39.256277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:39.386425Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:39.386425Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:14286b5e6b2c3b8002fe186d04607a802b8f77445d4048c2d3a3dc72a5467294","observation_id":"39ae2ef2-9b3e-4bc1-863d-8250ec475dc4","resolution":{"observed_at":"2026-08-02T19:46:39.386425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:39.554440Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:39.554440Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:731b2ce2d6acd0e44aae2fcfdafb88e1a3fc23bd05cf67313c06726945386021","observation_id":"c93db0da-2e08-4feb-9c4f-82d8797132c4","resolution":{"observed_at":"2026-08-02T19:46:39.554440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:39.667948Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:39.667948Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:57e219a4142311c60c7a557b79824c0cc84b6b769fa736a2d96fa4325508de26","observation_id":"af41fcfa-bb69-4fea-b1d8-ac4cc1f39f1f","resolution":{"observed_at":"2026-08-02T19:46:39.667948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:39.785839Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:39.785839Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:40d89f4865672d8d07d33ebc24df4ff0b62402f694f5626d903c32f195611273","observation_id":"9730cfca-09d8-439a-bc87-3460fdda7ef3","resolution":{"observed_at":"2026-08-02T19:46:39.785839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02592","last_updated":"2025-07-03T12:59:07Z","snapshot_observed_at":"2026-08-06T06:12:37.171726Z","submitted_at":"2025-07-03T12:59:07Z","title":"WebSailor: Navigating Super-human Reasoning for Web Agent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02592","snapshot_observed_at":"2026-08-02T19:46:39.940241Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:39.940241Z"},"links":{"cited_paper":"/paper/2507.02592","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:735312a612b31e2f7099176de4ca7fa2ae7a7af45de209579185ff2683a77618","observation_id":"340d4b9f-926c-42a5-bdab-7b28a47c9e56","resolution":{"observed_at":"2026-08-02T19:46:39.940241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:40.107847Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:40.107847Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:2dd6e81e99fb38e0d515477ba7753ea8f3a209e8d060d4453ff957127ab466b8","observation_id":"871b0b53-c439-4836-9c53-6d3cc4a184e2","resolution":{"observed_at":"2026-08-02T19:46:40.107847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:40.172725Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:40.172725Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:49b1cd2171692f84d2ee3e2b1e64223aa0df2024b62218398f0deee6e009105d","observation_id":"9db9e177-1246-49cf-be9b-c11220de2768","resolution":{"observed_at":"2026-08-02T19:46:40.172725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-02T19:46:40.178894Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:40.178894Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:5d86e3a28fd61786709ef6443db6ea74afac958b179a4338198b683d7e30c8fd","observation_id":"248a9c5a-b85a-4d20-908d-f341b6649ea1","resolution":{"observed_at":"2026-08-02T19:46:40.178894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2512.02556","last_updated":"2025-12-02T09:25:14Z","snapshot_observed_at":"2026-07-31T23:49:25.878472Z","submitted_at":"2025-12-02T09:25:14Z","title":"DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.02556","snapshot_observed_at":"2026-08-02T19:46:40.280720Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:40.280720Z"},"links":{"cited_paper":"/paper/2512.02556","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:88d7fd6f9ae912b2139e8ad156c7a8be7eefd2b24218489aae7930b244a7b4b4","observation_id":"33a08749-7934-4b46-b528-c8289ccfa1d2","resolution":{"observed_at":"2026-08-02T19:46:40.280720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:40.446778Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:40.446778Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:50a7fbe2cef1e9f680fb5d3fc87aeb6945df227e189f0bb314ab3d9fdf40dfe0","observation_id":"3f6e35e2-b34f-47d6-8382-7db5c1905292","resolution":{"observed_at":"2026-08-02T19:46:40.446778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.03688","last_updated":"2025-10-04T03:54:18Z","snapshot_observed_at":"2026-08-06T20:36:41.418114Z","submitted_at":"2023-08-07T16:08:11Z","title":"AgentBench: Evaluating LLMs as Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.03688","snapshot_observed_at":"2026-08-02T19:46:40.693646Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:40.693646Z"},"links":{"cited_paper":"/paper/2308.03688","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:5ae4167e7e4d766f86832f9e1291aa33753f1b85c9c12fbf270a1d6a9a55c07b","observation_id":"db980910-35c8-4186-ba72-6a240f292686","resolution":{"observed_at":"2026-08-02T19:46:40.693646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05960","last_updated":"2023-08-11T06:37:54Z","snapshot_observed_at":"2026-08-02T20:56:26.359474Z","submitted_at":"2023-08-11T06:37:54Z","title":"BOLAA: Benchmarking and Orchestrating LLM-augmented Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05960","snapshot_observed_at":"2026-08-02T19:46:40.814191Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:40.814191Z"},"links":{"cited_paper":"/paper/2308.05960","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:3385e2d157792aa059be7ceb43aea05c7afddd99981733dce388074508e574f2","observation_id":"388fdae6-b527-463c-ae60-00610c7b0d83","resolution":{"observed_at":"2026-08-02T19:46:40.814191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:40.922664Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:40.922664Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:e466e19dc671dbc9376c9d334c920dd56c51b14c5ef2bf7a77f707bd43778240","observation_id":"430068e1-d579-4802-8706-d66635094b2d","resolution":{"observed_at":"2026-08-02T19:46:40.922664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1906.02916","last_updated":"2019-06-30T22:30:19Z","snapshot_observed_at":"2026-07-06T07:58:39.570955Z","submitted_at":"2019-06-07T06:22:17Z","title":"Multi-hop Reading Comprehension through Question Decomposition and Rescoring","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1906.02916","snapshot_observed_at":"2026-08-02T19:46:41.018930Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.018930Z"},"links":{"cited_paper":"/paper/1906.02916","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:6543337f9eb8d7abcad4e2af4557a8d11419a0b3229f757acfbdb21107201840","observation_id":"f8fd960b-9593-4c2b-b646-e98ccc7bdfb0","resolution":{"observed_at":"2026-08-02T19:46:41.018930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:41.105225Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.105225Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:69d0803b1d283cb4a50692f3f9061b76d862d04b3c492ee854f8acee768f91f9","observation_id":"f26f8ccc-89ab-4a3c-a48d-d7c4db193f56","resolution":{"observed_at":"2026-08-02T19:46:41.105225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-08-02T19:46:41.256849Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.256849Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:6bbe7ff2b5eba915374fc7d24c3db84a1a703c9527f608e824f9d8483cafda53","observation_id":"2cbea9c6-03be-4beb-b4d6-cc1e4b085cf0","resolution":{"observed_at":"2026-08-02T19:46:41.256849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:41.344453Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.344453Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:d63ab4a69d4cda3b174837c59b60d046e310092728e81576ca11b14f396c6799","observation_id":"f3f271ab-d2ac-4aae-9b8c-2dea2bd704cf","resolution":{"observed_at":"2026-08-02T19:46:41.344453Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14249","snapshot_observed_at":"2026-08-02T19:46:41.429321Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.429321Z"},"links":{"cited_paper":"/paper/2501.14249","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:57108eb6b347c975235e2115abcfee923806363db814eb8651762eb0675a88d4","observation_id":"9a37c819-12ac-46bf-8d1e-4d0f9e66a32a","resolution":{"observed_at":"2026-08-02T19:46:41.429321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.07000","last_updated":"2019-10-15T18:44:47Z","snapshot_observed_at":"2026-07-06T08:29:50.317685Z","submitted_at":"2019-10-15T18:44:47Z","title":"Answering Complex Open-domain Questions Through Iterative Query Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.07000","snapshot_observed_at":"2026-08-02T19:46:41.514302Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.514302Z"},"links":{"cited_paper":"/paper/1910.07000","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:f7af6fa7495c47eec45db6c83445da1c5f85d8f643ab1d5a438696467104bc58","observation_id":"5de849f5-364d-40f1-a26a-0ca3e6c122dc","resolution":{"observed_at":"2026-08-02T19:46:41.514302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:41.599213Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.599213Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:6c4a7a7b3e851fc254efc7bc45f5af0349dbb3897a692e5ae5f1f7679ed42d8d","observation_id":"46211042-698b-4f5a-b376-598049e1d4a1","resolution":{"observed_at":"2026-08-02T19:46:41.599213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:41.697929Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.697929Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:0380080b46066c87dd1c52fb7d42c81aaa63840235aa1c6e5aca86c81d595999","observation_id":"e9aed41f-1f31-408b-b7d1-c73ffc37b8e3","resolution":{"observed_at":"2026-08-02T19:46:41.697929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16789","last_updated":"2023-10-03T14:45:48Z","snapshot_observed_at":"2026-07-06T16:00:46.542753Z","submitted_at":"2023-07-31T15:56:53Z","title":"ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.16789","snapshot_observed_at":"2026-08-02T19:46:41.757885Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.757885Z"},"links":{"cited_paper":"/paper/2307.16789","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:7a177f8155e84bd34b29aec66456b582057042e32c4009e09b1345091d5c048d","observation_id":"4ced405d-3d45-4977-aff5-f6fa05c79613","resolution":{"observed_at":"2026-08-02T19:46:41.757885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:41.835899Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.835899Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:ce2593f54422bb0f5a5a6760fa2dcdd79c6388bc407e0d83b1db8861a1abd039","observation_id":"1c76e1a7-3b37-4981-8107-44b1aca2ed47","resolution":{"observed_at":"2026-08-02T19:46:41.835899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:41.955351Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.955351Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:9c766d006a26cf3640137a9ebc4ebdc8f393b9f3701f4da9520f359fce2dc3c1","observation_id":"024791d7-997c-463a-b70d-35bd3f3838a6","resolution":{"observed_at":"2026-08-02T19:46:41.955351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-02T19:46:42.270579Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:42.270579Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:fda17d2c0240953dd0edc9a62ccf2a5642688d282998d09b967b59fdfd4a3092","observation_id":"be8de349-57e1-4fde-b7a0-5b807f2be1fb","resolution":{"observed_at":"2026-08-02T19:46:42.270579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:42.364309Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:42.364309Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:e12a810d7b3fa5bac8c298c9c9a12ddad2cb6993d60d81c31ba37499d40a0544","observation_id":"9b842d84-b5c3-41ae-82ee-0b63e91fdce9","resolution":{"observed_at":"2026-08-02T19:46:42.364309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:42.467286Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:42.467286Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:1012207cb424a5e368d08c1d17b5fe14aad9dda2b9615c90c39e0e9509f4a988","observation_id":"14e550b2-b66f-4afd-92fe-1265e4c8186e","resolution":{"observed_at":"2026-08-02T19:46:42.467286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:42.671546Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:42.671546Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:9b0c3c75ea527985abe8f37b5b7c4a883f0a4cb093f560c42361bc16939dcca4","observation_id":"194e607f-1bfa-435e-b640-fa6f2d2e077d","resolution":{"observed_at":"2026-08-02T19:46:42.671546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.24701","last_updated":"2026-05-18T04:10:32Z","snapshot_observed_at":"2026-07-06T22:34:18.297603Z","submitted_at":"2025-10-28T17:53:02Z","title":"Tongyi DeepResearch Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.24701","snapshot_observed_at":"2026-08-02T19:46:42.737492Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:42.737492Z"},"links":{"cited_paper":"/paper/2510.24701","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:f88386c9f5debb78c1707a5b956a6c834f9586523266f8a9e849ff9a040b3c0d","observation_id":"8e4d0d81-42c9-43eb-b59f-972485bfa129","resolution":{"observed_at":"2026-08-02T19:46:42.737492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08663","last_updated":"2021-10-21T01:18:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-04-17T23:29:55Z","title":"BEIR: A Heterogenous Benchmark for Zero-shot Evaluation of Information Retrieval Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08663","snapshot_observed_at":"2026-08-02T19:46:42.816248Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:42.816248Z"},"links":{"cited_paper":"/paper/2104.08663","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:146ff6d656e5dce94e9ed001b7a3b7bbba053049b0f66ca507f4e2e986a44b7b","observation_id":"cb218cc8-5bcb-41a2-b821-2924ff8349f6","resolution":{"observed_at":"2026-08-02T19:46:42.816248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:42.943706Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:42.943706Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:2461ec99e63dedca3541e8b96ee01d513bc23a49c6185865efa39691fd3e302f","observation_id":"4455723c-abec-4331-b85c-a11ed65bcc32","resolution":{"observed_at":"2026-08-02T19:46:42.943706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14283","last_updated":"2024-07-22T10:01:49Z","snapshot_observed_at":"2026-07-06T18:34:15.291938Z","submitted_at":"2024-06-20T13:08:09Z","title":"Q*: Improving Multi-step Reasoning for LLMs with Deliberative Planning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14283","snapshot_observed_at":"2026-08-02T19:46:43.187057Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:43.187057Z"},"links":{"cited_paper":"/paper/2406.14283","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:35cf0774ef5498c01309dd01f11456560d1a29dab7a05d3bd0ac2c63818d2b39","observation_id":"ec8c5023-e6f9-4d50-ba88-74a9af2fd8d5","resolution":{"observed_at":"2026-08-02T19:46:43.187057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:43.296106Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:43.296106Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:dd7d3edc6a194af7d0bb050d9b6784bf74bde26c8087499862bd809658139f51","observation_id":"72154966-bbc2-4ab0-8c5e-e75ee3ffd7ac","resolution":{"observed_at":"2026-08-02T19:46:43.296106Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:43.393593Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:43.393593Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:fed2d1e3cbe1ed343b5562d07123b9b3fca82253742f0a1c321647af0baf2fe6","observation_id":"8985199e-0734-4f7b-8760-92c7a056981a","resolution":{"observed_at":"2026-08-02T19:46:43.393593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:43.747753Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:43.747753Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:3a51333e199baa86edc9a1463b7782645e2bfc0d2a91e5a8e506c8bdcdb149fd","observation_id":"b13dee8c-ceb8-49f4-a79c-a5aa64260c06","resolution":{"observed_at":"2026-08-02T19:46:43.747753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:43.902448Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:43.902448Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:e9483bfc227ea482c3fe221db71b6fe32948069cf5f02528601ceecdd9f5bdd8","observation_id":"cbae67e1-00cb-44b5-bc47-777a4f6758ab","resolution":{"observed_at":"2026-08-02T19:46:43.902448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:43.085349Z","title":"Transactions of the Association for Computational Linguistics10 (2022), 539–554","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:43.085349Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:5dce12c6db258198786e6d1e0a44c4877139f3a6b11b6a276d124f66207cc56b","observation_id":"da714e4f-84ba-468d-8c2c-8a1d9fd0eea3","resolution":{"observed_at":"2026-08-02T19:46:43.085349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.12756","last_updated":"2021-02-19T22:15:03Z","snapshot_observed_at":"2026-08-05T01:57:43.963392Z","submitted_at":"2020-09-27T06:12:29Z","title":"Answering Complex Open-Domain Questions with Multi-Hop Dense Retrieval","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.12756","snapshot_observed_at":"2026-08-02T19:46:44.108789Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:44.108789Z"},"links":{"cited_paper":"/paper/2009.12756","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:55cb33da73ce0ef84c3ea232bb2854c147eb5fa36908dada7b889358129f04b6","observation_id":"0ab0762c-ce21-45d4-87ef-212e6a27a951","resolution":{"observed_at":"2026-08-02T19:46:44.108789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.12594","last_updated":"2025-06-14T18:19:05Z","snapshot_observed_at":"2026-08-07T00:43:03.992583Z","submitted_at":"2025-06-14T18:19:05Z","title":"A Comprehensive Survey of Deep Research: Systems, Methodologies, and Applications","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.12594","snapshot_observed_at":"2026-08-02T19:46:44.219868Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:44.219868Z"},"links":{"cited_paper":"/paper/2506.12594","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:3f1d81c17f1689b172f110a2aacdae9c445038ae4699726167a167549689623c","observation_id":"1033bced-947c-47a6-bf01-8701ad8d9315","resolution":{"observed_at":"2026-08-02T19:46:44.219868Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-02T19:46:44.386944Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:44.386944Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:c538e54e683720583da8e9bfd7f3aa33e89085f41ada14061283aa7510298767","observation_id":"10e73ecf-2106-417e-9df5-69b207794805","resolution":{"observed_at":"2026-08-02T19:46:44.386944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12516","last_updated":"2025-04-16T22:27:45Z","snapshot_observed_at":"2026-08-03T00:43:33.338074Z","submitted_at":"2025-04-16T22:27:45Z","title":"BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.12516","snapshot_observed_at":"2026-08-02T19:46:43.536223Z","title":"arXiv preprint arXiv:2504.12516(2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:43.536223Z"},"links":{"cited_paper":"/paper/2504.12516","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:7485fd080aeb8a6d7e758b1408e320ff518cb2ee85bae47f399e16557253e008","observation_id":"226996e0-590f-4bf1-9419-74cc107b499c","resolution":{"observed_at":"2026-08-02T19:46:43.536223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:44.638140Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:44.638140Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:a759ddfe8fef239c157e942a258062529cbed4fc9e1504b444f8bb6196fe3aae","observation_id":"79ab8a80-1c1a-416d-84ed-d494b9932b51","resolution":{"observed_at":"2026-08-02T19:46:44.638140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18892","last_updated":"2025-08-06T08:42:32Z","snapshot_observed_at":"2026-07-06T20:57:57.039376Z","submitted_at":"2025-03-24T17:06:10Z","title":"SimpleRL-Zoo: Investigating and Taming Zero Reinforcement Learning for Open Base Models in the Wild","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.18892","snapshot_observed_at":"2026-08-02T19:46:44.793381Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:44.793381Z"},"links":{"cited_paper":"/paper/2503.18892","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:9d1b57566a00a946946aa74ca6c705959607abf86e9e077c2859775bf859c6c4","observation_id":"397e59d0-32dd-430e-9110-8914bce5e543","resolution":{"observed_at":"2026-08-02T19:46:44.793381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:44.020730Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:44.020730Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:ac1c2ce375f53ac6a40ceeabf3b3e94a9dcbc2fd03742947479e17825913ed0a","observation_id":"b0d79acf-217c-4b27-b4cb-8f5a559336fc","resolution":{"observed_at":"2026-08-02T19:46:44.020730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:45.179215Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:45.179215Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:d9822d13be50f4540a3220715961ad4c7b2ba84075c69d50386ebedcb8dfcb55","observation_id":"879ba7a4-da55-45d6-bc44-375fdeb568ef","resolution":{"observed_at":"2026-08-02T19:46:45.179215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:44.486904Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:44.486904Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:1fc8fde0d9da3bb2fd1ba538164e5eb7126d944846fac30d3a9f275450d4a5fb","observation_id":"61f8f9d4-919b-4596-b8f9-348f88ed5826","resolution":{"observed_at":"2026-08-02T19:46:44.486904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:44.913515Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:44.913515Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:616a5920ba42ee2c47f415520c5b318ee8d4ec0174ae614171f7a493fc833646","observation_id":"593453fc-e6a8-49c2-8bc8-81c0ecfecde9","resolution":{"observed_at":"2026-08-02T19:46:44.913515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-02T19:46:42.101108Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:42.101108Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:ad0de8a34b796b5a4ce5aaf158849c0b9bbe42346e7115ed4d309d6bcd0ebae8","observation_id":"71bda43f-0691-43c4-b9a2-8001e3bd0924","resolution":{"observed_at":"2026-08-02T19:46:42.101108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.08124","last_updated":"2020-07-16T05:52:16Z","snapshot_observed_at":"2026-07-06T09:38:41.713097Z","submitted_at":"2020-07-16T05:52:16Z","title":"LogiQA: A Challenge Dataset for Machine Reading Comprehension with Logical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.08124","snapshot_observed_at":"2026-08-02T19:46:40.534368Z","title":null,"venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:40.534368Z"},"links":{"cited_paper":"/paper/2007.08124","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:9c6800599705428a66d609a713d01480da653882e0fbab6382c83682a60c9711","observation_id":"a882b5d5-f23b-4971-8a56-0c857da2e806","resolution":{"observed_at":"2026-08-02T19:46:40.534368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.17517","last_updated":"2023-03-08T16:47:46Z","snapshot_observed_at":"2026-08-04T16:15:25.064317Z","submitted_at":"2022-10-31T17:41:26Z","title":"Lila: A Unified Benchmark for Mathematical Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.17517","snapshot_observed_at":"2026-08-02T19:46:41.170740Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:41.170740Z"},"links":{"cited_paper":"/paper/2210.17517","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:4cd2be35db86d4d8e15677861efcbc51df1cc890eba91a4d0fc581a3546c316c","observation_id":"fa47cefa-cb68-4bee-b481-878d613f174a","resolution":{"observed_at":"2026-08-02T19:46:41.170740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T19:46:36.140715Z","title":"InProceedings of the 2023 Conference on Empirical Methods in Natural Language Processing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:36.140715Z"},"links":{"citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:77937c957b0630cf6b3ffcc873a03ac2aaaa540a6cc60ec4e3ef4fe1b4074f50","observation_id":"abb0b937-b17d-4f67-9eab-1623a1745b23","resolution":{"observed_at":"2026-08-02T19:46:36.140715Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14924","last_updated":"2024-09-23T11:20:20Z","snapshot_observed_at":"2026-07-06T19:20:23.589464Z","submitted_at":"2024-09-23T11:20:20Z","title":"Retrieval Augmented Generation (RAG) and Beyond: A Comprehensive Survey on How to Make your LLMs use External Data More Wisely","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.14924","snapshot_observed_at":"2026-08-02T19:46:45.069746Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:45.069746Z"},"links":{"cited_paper":"/paper/2409.14924","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:6e64b9c6fac60ae1ce5ab419842b9ee6069ff70f57004667ce50ed1cf345e212","observation_id":"3b201037-1f8b-437f-995d-4af0f1553901","resolution":{"observed_at":"2026-08-02T19:46:45.069746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03267","last_updated":"2026-05-01T23:55:43Z","snapshot_observed_at":"2026-08-02T10:52:10.211700Z","submitted_at":"2025-12-19T07:05:38Z","title":"OpenAI GPT-5 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.03267","snapshot_observed_at":"2026-08-02T19:46:42.554524Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-02T19:46:42.554524Z"},"links":{"cited_paper":"/paper/2601.03267","citing_paper":"/paper/2603.01152"},"observation_digest":"sha256:59f6017d0a80027643ef432a6117380386faa356d05fd555b79585dd5f01ba39","observation_id":"94a22af8-29db-44d8-a2a3-916a7ba903db","resolution":{"observed_at":"2026-08-02T19:46:42.554524Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2603.01152","last_updated":"2026-06-20T09:26:23Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-07T06:59:58.157969Z","submitted_at":"2026-03-01T15:36:10Z","title":"DeepResearch-9K: A Challenging Benchmark Dataset of Deep-Research Agent"},"reference_resolution":{"displayed":77,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":77,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":77},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 77 of 77 outbound references and 5 inbound Pith citation observations for arXiv:2603.01152."}