{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XNHZSR66M72PVT44AQYFXWYBZT","short_pith_number":"pith:XNHZSR66","schema_version":"1.0","canonical_sha256":"bb4f9947de67f4facf9c04305bdb01ccf4e414732a896532ade2664af51244f8","source":{"kind":"arxiv","id":"2505.15117","version":1},"attestation_state":"computed","paper":{"title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Bowen Jin, Jiawei Han, Jinsung Yoon, Priyanka Kargupta, Sercan O. Arik","submitted_at":"2025-05-21T05:09:43Z","abstract_excerpt":"Reinforcement learning (RL) has demonstrated strong potential in training large language models (LLMs) capable of complex reasoning for real-world problem solving. More recently, RL has been leveraged to create sophisticated LLM-based search agents that adeptly combine reasoning with search engine use. While the use of RL for training search agents is promising, the optimal design of such agents remains not fully understood. In particular, key factors -- such as (1) reward formulation, (2) the choice and characteristics of the underlying LLM, and (3) the role of the search engine in the RL pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.15117","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-21T05:09:43Z","cross_cats_sorted":["cs.AI","cs.IR"],"title_canon_sha256":"c39a647646889582232c9825224d1ad2031c6cfa8748fa3a418bc5490b48b247","abstract_canon_sha256":"4f026fc4f469c2c2a45f5004c12d1a552060bb280cba834fdac7d02f7ae6b857"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:41.000086Z","signature_b64":"ZKwRjiJHYIduu70cm1A/+gL1rZzKcBENkHxsnMcLI7bE6V26KLZY21StrlkKNhm27shPseFbA3pU/buGUgxiAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb4f9947de67f4facf9c04305bdb01ccf4e414732a896532ade2664af51244f8","last_reissued_at":"2026-07-05T11:06:40.999624Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:40.999624Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Bowen Jin, Jiawei Han, Jinsung Yoon, Priyanka Kargupta, Sercan O. Arik","submitted_at":"2025-05-21T05:09:43Z","abstract_excerpt":"Reinforcement learning (RL) has demonstrated strong potential in training large language models (LLMs) capable of complex reasoning for real-world problem solving. More recently, RL has been leveraged to create sophisticated LLM-based search agents that adeptly combine reasoning with search engine use. While the use of RL for training search agents is promising, the optimal design of such agents remains not fully understood. In particular, key factors -- such as (1) reward formulation, (2) the choice and characteristics of the underlying LLM, and (3) the role of the search engine in the RL pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.15117","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.15117/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.15117","created_at":"2026-07-05T11:06:40.999679+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.15117v1","created_at":"2026-07-05T11:06:40.999679+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.15117","created_at":"2026-07-05T11:06:40.999679+00:00"},{"alias_kind":"pith_short_12","alias_value":"XNHZSR66M72P","created_at":"2026-07-05T11:06:40.999679+00:00"},{"alias_kind":"pith_short_16","alias_value":"XNHZSR66M72PVT44","created_at":"2026-07-05T11:06:40.999679+00:00"},{"alias_kind":"pith_short_8","alias_value":"XNHZSR66","created_at":"2026-07-05T11:06:40.999679+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01248","citing_title":"$S^3$-R1: Learning to Retrieve and Answer Step-by-Step with Synthetic Data","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26122","citing_title":"DocArena: Turning Raw Documents into Controllable Training Environments for Document Search Agents","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2510.00861","citing_title":"Erase to Improve: Erasable Reinforcement Learning for Search-Augmented LLMs","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06285","citing_title":"LatentRAG: Latent Reasoning and Retrieval for Efficient Agentic RAG","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01399","citing_title":"Verbal-R3: Verbal Reranker as the Missing Bridge between Retrieval and Reasoning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01248","citing_title":"$S^3$-R1: Learning to Retrieve and Answer Step-by-Step with Synthetic Data","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XNHZSR66M72PVT44AQYFXWYBZT","json":"https://pith.science/pith/XNHZSR66M72PVT44AQYFXWYBZT.json","graph_json":"https://pith.science/api/pith-number/XNHZSR66M72PVT44AQYFXWYBZT/graph.json","events_json":"https://pith.science/api/pith-number/XNHZSR66M72PVT44AQYFXWYBZT/events.json","paper":"https://pith.science/paper/XNHZSR66"},"agent_actions":{"view_html":"https://pith.science/pith/XNHZSR66M72PVT44AQYFXWYBZT","download_json":"https://pith.science/pith/XNHZSR66M72PVT44AQYFXWYBZT.json","view_paper":"https://pith.science/paper/XNHZSR66","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.15117&json=true","fetch_graph":"https://pith.science/api/pith-number/XNHZSR66M72PVT44AQYFXWYBZT/graph.json","fetch_events":"https://pith.science/api/pith-number/XNHZSR66M72PVT44AQYFXWYBZT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XNHZSR66M72PVT44AQYFXWYBZT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XNHZSR66M72PVT44AQYFXWYBZT/action/storage_attestation","attest_author":"https://pith.science/pith/XNHZSR66M72PVT44AQYFXWYBZT/action/author_attestation","sign_citation":"https://pith.science/pith/XNHZSR66M72PVT44AQYFXWYBZT/action/citation_signature","submit_replication":"https://pith.science/pith/XNHZSR66M72PVT44AQYFXWYBZT/action/replication_record"}},"created_at":"2026-07-05T11:06:40.999679+00:00","updated_at":"2026-07-05T11:06:40.999679+00:00"}