{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DKQPLDWICYA7H6YRJZHARNKIVI","short_pith_number":"pith:DKQPLDWI","schema_version":"1.0","canonical_sha256":"1aa0f58ec81601f3fb114e4e08b548aa048ce40c33a4dcdc398cf7cccce98efe","source":{"kind":"arxiv","id":"2502.01142","version":2},"attestation_state":"computed","paper":{"title":"DeepRAG: Thinking to Retrieve Step by Step for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.IR"],"primary_cat":"cs.AI","authors_text":"Chunlei Xin, Fandong Meng, Hongyu Lin, Jiali Zeng, Jie Zhou, Le Sun, Xianpei Han, Xinyan Guan, Yaojie Lu","submitted_at":"2025-02-03T08:22:45Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable reasoning capabilities, while their practical applications are limited by severe factual hallucinations due to limitations in the timeliness, accuracy, and comprehensiveness of their parametric knowledge. Meanwhile, enhancing retrieval-augmented generation (RAG) with reasoning remains challenging due to ineffective task decomposition and redundant retrieval, which can introduce noise and degrade response quality. In this paper, we propose DeepRAG, a framework that models retrieval-augmented reasoning as a Markov Decision Process (MDP), enablin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01142","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-02-03T08:22:45Z","cross_cats_sorted":["cs.CL","cs.IR"],"title_canon_sha256":"5b3021e52ce1a6ce6c379d229649c6c50327249f66c2e263f64fd508dbdd54de","abstract_canon_sha256":"ac3a7479fc1728b4840964102ed5691f4b482cb8cb9bcf9d2219c17bf45d4125"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:40.476530Z","signature_b64":"ohJAMkgaj+8PKi4kmUMeTGoN/dMk6GDQDmMgoE4MpMvcJkCu8xRdw8NusrYL6qD/FrGzQK0oCF34BG9w6ZGEAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1aa0f58ec81601f3fb114e4e08b548aa048ce40c33a4dcdc398cf7cccce98efe","last_reissued_at":"2026-07-05T11:17:40.475799Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:40.475799Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DeepRAG: Thinking to Retrieve Step by Step for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.IR"],"primary_cat":"cs.AI","authors_text":"Chunlei Xin, Fandong Meng, Hongyu Lin, Jiali Zeng, Jie Zhou, Le Sun, Xianpei Han, Xinyan Guan, Yaojie Lu","submitted_at":"2025-02-03T08:22:45Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable reasoning capabilities, while their practical applications are limited by severe factual hallucinations due to limitations in the timeliness, accuracy, and comprehensiveness of their parametric knowledge. Meanwhile, enhancing retrieval-augmented generation (RAG) with reasoning remains challenging due to ineffective task decomposition and redundant retrieval, which can introduce noise and degrade response quality. In this paper, we propose DeepRAG, a framework that models retrieval-augmented reasoning as a Markov Decision Process (MDP), enablin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01142","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01142/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01142","created_at":"2026-07-05T11:17:40.475915+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01142v2","created_at":"2026-07-05T11:17:40.475915+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01142","created_at":"2026-07-05T11:17:40.475915+00:00"},{"alias_kind":"pith_short_12","alias_value":"DKQPLDWICYA7","created_at":"2026-07-05T11:17:40.475915+00:00"},{"alias_kind":"pith_short_16","alias_value":"DKQPLDWICYA7H6YR","created_at":"2026-07-05T11:17:40.475915+00:00"},{"alias_kind":"pith_short_8","alias_value":"DKQPLDWI","created_at":"2026-07-05T11:17:40.475915+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07299","citing_title":"DuMate-DeepResearch: An Auditable Multi-Agent System with Recursive Search and Rubric-Grounded Reasoning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21460","citing_title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2505.22095","citing_title":"Mixture-of-Retrieval Experts for Reasoning-Guided Multimodal Knowledge Exploitation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2510.07794","citing_title":"HiPRAG: Hierarchical Process Rewards for Efficient Agentic Retrieval Augmented Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2601.12538","citing_title":"Agentic Reasoning for Large Language Models","ref_index":257,"is_internal_anchor":false},{"citing_arxiv_id":"2504.21776","citing_title":"WebThinker: Empowering Large Reasoning Models with Deep Research Capability","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04969","citing_title":"MG$^2$-RAG: Multi-Granularity Graph for Multimodal Retrieval-Augmented Generation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":225,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06416","citing_title":"MiA-Signature: Approximating Global Activation for Long-Context Understanding","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05952","citing_title":"Towards Trustworthy Report Generation: A Deep Research Agent with Progressive Confidence Estimation and Calibration","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19144","citing_title":"ReflectMT: Internalizing Reflection for Efficient and High-Quality Machine Translation","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DKQPLDWICYA7H6YRJZHARNKIVI","json":"https://pith.science/pith/DKQPLDWICYA7H6YRJZHARNKIVI.json","graph_json":"https://pith.science/api/pith-number/DKQPLDWICYA7H6YRJZHARNKIVI/graph.json","events_json":"https://pith.science/api/pith-number/DKQPLDWICYA7H6YRJZHARNKIVI/events.json","paper":"https://pith.science/paper/DKQPLDWI"},"agent_actions":{"view_html":"https://pith.science/pith/DKQPLDWICYA7H6YRJZHARNKIVI","download_json":"https://pith.science/pith/DKQPLDWICYA7H6YRJZHARNKIVI.json","view_paper":"https://pith.science/paper/DKQPLDWI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01142&json=true","fetch_graph":"https://pith.science/api/pith-number/DKQPLDWICYA7H6YRJZHARNKIVI/graph.json","fetch_events":"https://pith.science/api/pith-number/DKQPLDWICYA7H6YRJZHARNKIVI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DKQPLDWICYA7H6YRJZHARNKIVI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DKQPLDWICYA7H6YRJZHARNKIVI/action/storage_attestation","attest_author":"https://pith.science/pith/DKQPLDWICYA7H6YRJZHARNKIVI/action/author_attestation","sign_citation":"https://pith.science/pith/DKQPLDWICYA7H6YRJZHARNKIVI/action/citation_signature","submit_replication":"https://pith.science/pith/DKQPLDWICYA7H6YRJZHARNKIVI/action/replication_record"}},"created_at":"2026-07-05T11:17:40.475915+00:00","updated_at":"2026-07-05T11:17:40.475915+00:00"}