{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NA2A4LRDZLIRGIW5C2PPVDAAAO","short_pith_number":"pith:NA2A4LRD","schema_version":"1.0","canonical_sha256":"68340e2e23cad11322dd169efa8c00039bffc4e2636a93651a7f6829eeeee7b7","source":{"kind":"arxiv","id":"2505.15872","version":2},"attestation_state":"computed","paper":{"title":"InfoDeepSeek: Benchmarking Agentic Information Seeking for Retrieval-Augmented Generation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Bo Chen, Jianghao Lin, Jiaqi Liu, Menghui Zhu, Ruiming Tang, Tong Wan, Weinan Zhang, Weiwen Liu, Yasheng Wang, Yong Yu, Yongzhao Xiao, Yunjia Xi, Zhuoying Ou","submitted_at":"2025-05-21T14:44:40Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) enhances large language models (LLMs) by grounding responses with retrieved information. As an emerging paradigm, Agentic RAG further enhances this process by introducing autonomous LLM agents into the information seeking process. However, existing benchmarks fall short in evaluating such systems, as they are confined to a static retrieval environment with a fixed, limited corpus} and simple queries that fail to elicit agentic behavior. Moreover, their evaluation protocols assess information seeking effectiveness by pre-defined gold sets of documents, makin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.15872","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.IR","submitted_at":"2025-05-21T14:44:40Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"71d4e02df8b8c9320122db3e54dd91c75e79dd6c7a420ab9a1c3c6f5fd0b4646","abstract_canon_sha256":"366e77752d60b5168c2e28cbcbe05a560819f58784f2cb4c33257f43144bec95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:21.687807Z","signature_b64":"8ytZeWKK5F4/MkBJn20NGZjpeOR6m1/xg0g8rzLR9BsJvCGorNKiRTzXvCF2ezS3x35Cbkgt7c6+aswbwqOqAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"68340e2e23cad11322dd169efa8c00039bffc4e2636a93651a7f6829eeeee7b7","last_reissued_at":"2026-07-05T11:08:21.687340Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:21.687340Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InfoDeepSeek: Benchmarking Agentic Information Seeking for Retrieval-Augmented Generation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Bo Chen, Jianghao Lin, Jiaqi Liu, Menghui Zhu, Ruiming Tang, Tong Wan, Weinan Zhang, Weiwen Liu, Yasheng Wang, Yong Yu, Yongzhao Xiao, Yunjia Xi, Zhuoying Ou","submitted_at":"2025-05-21T14:44:40Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) enhances large language models (LLMs) by grounding responses with retrieved information. As an emerging paradigm, Agentic RAG further enhances this process by introducing autonomous LLM agents into the information seeking process. However, existing benchmarks fall short in evaluating such systems, as they are confined to a static retrieval environment with a fixed, limited corpus} and simple queries that fail to elicit agentic behavior. Moreover, their evaluation protocols assess information seeking effectiveness by pre-defined gold sets of documents, makin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.15872","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.15872/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.15872","created_at":"2026-07-05T11:08:21.687400+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.15872v2","created_at":"2026-07-05T11:08:21.687400+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.15872","created_at":"2026-07-05T11:08:21.687400+00:00"},{"alias_kind":"pith_short_12","alias_value":"NA2A4LRDZLIR","created_at":"2026-07-05T11:08:21.687400+00:00"},{"alias_kind":"pith_short_16","alias_value":"NA2A4LRDZLIRGIW5","created_at":"2026-07-05T11:08:21.687400+00:00"},{"alias_kind":"pith_short_8","alias_value":"NA2A4LRD","created_at":"2026-07-05T11:08:21.687400+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12191","citing_title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11926","citing_title":"Toward Generalist Autonomous Research via Hypothesis-Tree Refinement","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25256","citing_title":"AutoResearchBench: Benchmarking AI Agents on Complex Scientific Literature Discovery","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18146","citing_title":"Modular Representation Compression: Adapting LLMs for Efficient and Effective Recommendations","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14896","citing_title":"Toward Agentic RAG for Ukrainian","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NA2A4LRDZLIRGIW5C2PPVDAAAO","json":"https://pith.science/pith/NA2A4LRDZLIRGIW5C2PPVDAAAO.json","graph_json":"https://pith.science/api/pith-number/NA2A4LRDZLIRGIW5C2PPVDAAAO/graph.json","events_json":"https://pith.science/api/pith-number/NA2A4LRDZLIRGIW5C2PPVDAAAO/events.json","paper":"https://pith.science/paper/NA2A4LRD"},"agent_actions":{"view_html":"https://pith.science/pith/NA2A4LRDZLIRGIW5C2PPVDAAAO","download_json":"https://pith.science/pith/NA2A4LRDZLIRGIW5C2PPVDAAAO.json","view_paper":"https://pith.science/paper/NA2A4LRD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.15872&json=true","fetch_graph":"https://pith.science/api/pith-number/NA2A4LRDZLIRGIW5C2PPVDAAAO/graph.json","fetch_events":"https://pith.science/api/pith-number/NA2A4LRDZLIRGIW5C2PPVDAAAO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NA2A4LRDZLIRGIW5C2PPVDAAAO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NA2A4LRDZLIRGIW5C2PPVDAAAO/action/storage_attestation","attest_author":"https://pith.science/pith/NA2A4LRDZLIRGIW5C2PPVDAAAO/action/author_attestation","sign_citation":"https://pith.science/pith/NA2A4LRDZLIRGIW5C2PPVDAAAO/action/citation_signature","submit_replication":"https://pith.science/pith/NA2A4LRDZLIRGIW5C2PPVDAAAO/action/replication_record"}},"created_at":"2026-07-05T11:08:21.687400+00:00","updated_at":"2026-07-05T11:08:21.687400+00:00"}