{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RTG2365K34Q2D2YGJTO773KCIA","short_pith_number":"pith:RTG2365K","schema_version":"1.0","canonical_sha256":"8ccdadfbaadf21a1eb064cddffed424008b38fa5e2021ee4e58882d91a5d82c7","source":{"kind":"arxiv","id":"2508.12752","version":1},"attestation_state":"computed","paper":{"title":"Deep Research: A Survey of Autonomous Research Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Huifeng Guo, Pengyue Jia, Wenlin Zhang, Xiangyu Zhao, Xiaopeng Li, Yichao Wang, Yingyi Zhang, Yong Liu","submitted_at":"2025-08-18T09:26:14Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has driven the development of agentic systems capable of autonomously performing complex tasks. Despite their impressive capabilities, LLMs remain constrained by their internal knowledge boundaries. To overcome these limitations, the paradigm of deep research has been proposed, wherein agents actively engage in planning, retrieval, and synthesis to generate comprehensive and faithful analytical reports grounded in web-based evidence. In this survey, we provide a systematic overview of the deep research pipeline, which comprises four core st"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.12752","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2025-08-18T09:26:14Z","cross_cats_sorted":[],"title_canon_sha256":"57de5f307292bccbb6fd1456035f729330935bb97042265bf5fa504d03660433","abstract_canon_sha256":"af7763c519fc7b0464d0d45c9e42ea42658dfd2f71a50f4d91574568069ad2c6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:55:26.511649Z","signature_b64":"fKcEgil6MMa8UAVH4GMLLw+U6gn02tFNrWJNAW5b80FXXAAw/OmPhNyePIzMEXmeq1LuWiDjZ7w6VGT9sgMuAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ccdadfbaadf21a1eb064cddffed424008b38fa5e2021ee4e58882d91a5d82c7","last_reissued_at":"2026-07-05T11:55:26.511180Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:55:26.511180Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Research: A Survey of Autonomous Research Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Huifeng Guo, Pengyue Jia, Wenlin Zhang, Xiangyu Zhao, Xiaopeng Li, Yichao Wang, Yingyi Zhang, Yong Liu","submitted_at":"2025-08-18T09:26:14Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has driven the development of agentic systems capable of autonomously performing complex tasks. Despite their impressive capabilities, LLMs remain constrained by their internal knowledge boundaries. To overcome these limitations, the paradigm of deep research has been proposed, wherein agents actively engage in planning, retrieval, and synthesis to generate comprehensive and faithful analytical reports grounded in web-based evidence. In this survey, we provide a systematic overview of the deep research pipeline, which comprises four core st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.12752","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.12752/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.12752","created_at":"2026-07-05T11:55:26.511243+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.12752v1","created_at":"2026-07-05T11:55:26.511243+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.12752","created_at":"2026-07-05T11:55:26.511243+00:00"},{"alias_kind":"pith_short_12","alias_value":"RTG2365K34Q2","created_at":"2026-07-05T11:55:26.511243+00:00"},{"alias_kind":"pith_short_16","alias_value":"RTG2365K34Q2D2YG","created_at":"2026-07-05T11:55:26.511243+00:00"},{"alias_kind":"pith_short_8","alias_value":"RTG2365K","created_at":"2026-07-05T11:55:26.511243+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06374","citing_title":"VaseMuseum: Digital Intelligent Museum for Ancient Greek Pottery","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25342","citing_title":"Lifelong In-Context Learning with Transformers Requires Parametric Forms of Attention","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12736","citing_title":"Benchmarking AI Agents for Addressing Scientific Challenges Across Scales","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08671","citing_title":"SkillHone: A Harness for Continual Agent Skill Evolution Through Persistent Decision History","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07299","citing_title":"DuMate-DeepResearch: An Auditable Multi-Agent System with Recursive Search and Rubric-Grounded Reasoning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05241","citing_title":"Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30755","citing_title":"Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24198","citing_title":"Rewarding the Scientific Process: Process-Level Reward Modeling for Agentic Data Analysis","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31478","citing_title":"One Reflection Is Not Enough: Self-Correcting Autonomous Research via Multi-Hypothesis Failure Attribution","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26835","citing_title":"Helicase: Uncertainty-Guided Supply Chain Knowledge Graph Construction with Autonomous Multi-Agent LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28683","citing_title":"VeriTrip: A Verifiable Benchmark for Travel Planning Agents over Unstructured Web Corpora","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06416","citing_title":"Unsupervised Skill Discovery for Agentic Data Analysis","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20997","citing_title":"BioInsight: Multi-Agent Orchestration for Interactive Biomedical Knowledge Discovery","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21468","citing_title":"MemOCR: Layout-Aware Visual Memory for Efficient Long-Horizon Reasoning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17830","citing_title":"Remembering More, Risking More: Longitudinal Safety Risks in Memory-Equipped LLM Agents","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13034","citing_title":"ViDR: Grounding Multimodal Deep Research Reports in Source Visual Evidence","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10530","citing_title":"Personalized Deep Research: A User-Centric Framework, Dataset, and Hybrid Evaluation for Knowledge Discovery","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24198","citing_title":"Rewarding the Scientific Process: Process-Level Reward Modeling for Agentic Data Analysis","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14116","citing_title":"TREX: Automating LLM Fine-tuning via Agent-Driven Tree-based Exploration","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15715","citing_title":"GTA-2: Benchmarking General Tool Agents from Atomic Tool-Use to Open-Ended Workflows","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17265","citing_title":"MemSearch-o1: Empowering Large Language Models with Reasoning-Aligned Memory Growth in Agentic Search","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RTG2365K34Q2D2YGJTO773KCIA","json":"https://pith.science/pith/RTG2365K34Q2D2YGJTO773KCIA.json","graph_json":"https://pith.science/api/pith-number/RTG2365K34Q2D2YGJTO773KCIA/graph.json","events_json":"https://pith.science/api/pith-number/RTG2365K34Q2D2YGJTO773KCIA/events.json","paper":"https://pith.science/paper/RTG2365K"},"agent_actions":{"view_html":"https://pith.science/pith/RTG2365K34Q2D2YGJTO773KCIA","download_json":"https://pith.science/pith/RTG2365K34Q2D2YGJTO773KCIA.json","view_paper":"https://pith.science/paper/RTG2365K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.12752&json=true","fetch_graph":"https://pith.science/api/pith-number/RTG2365K34Q2D2YGJTO773KCIA/graph.json","fetch_events":"https://pith.science/api/pith-number/RTG2365K34Q2D2YGJTO773KCIA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RTG2365K34Q2D2YGJTO773KCIA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RTG2365K34Q2D2YGJTO773KCIA/action/storage_attestation","attest_author":"https://pith.science/pith/RTG2365K34Q2D2YGJTO773KCIA/action/author_attestation","sign_citation":"https://pith.science/pith/RTG2365K34Q2D2YGJTO773KCIA/action/citation_signature","submit_replication":"https://pith.science/pith/RTG2365K34Q2D2YGJTO773KCIA/action/replication_record"}},"created_at":"2026-07-05T11:55:26.511243+00:00","updated_at":"2026-07-05T11:55:26.511243+00:00"}