{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IRBHERLCQX2MXVNGWLQMCDHRGT","short_pith_number":"pith:IRBHERLC","schema_version":"1.0","canonical_sha256":"444272456285f4cbd5a6b2e0c10cf134e101a650adb85cadd74a5d0780e9d7d8","source":{"kind":"arxiv","id":"2508.10874","version":1},"attestation_state":"computed","paper":{"title":"SSRL: Self-Search Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bingning Wang, Bowen Zhou, Che Jiang, Cheng Huang, Gang Chen, Heng Zhou, Kaiyan Zhang, Lei Bai, Li Kang, Ning Ding, Xinwei Long, Xuekai Zhu, Yanxu Chen, Yuchen Fan, Yuchen Zhang, Yu Fu, Yuxin Zuo, Zhizhou He","submitted_at":"2025-08-14T17:46:01Z","abstract_excerpt":"We investigate the potential of large language models (LLMs) to serve as efficient simulators for agentic search tasks in reinforcement learning (RL), thereby reducing dependence on costly interactions with external search engines. To this end, we first quantify the intrinsic search capability of LLMs via structured prompting and repeated sampling, which we term Self-Search. Our results reveal that LLMs exhibit strong scaling behavior with respect to the inference budget, achieving high pass@k on question-answering benchmarks, including the challenging BrowseComp task. Building on these observ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.10874","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-08-14T17:46:01Z","cross_cats_sorted":[],"title_canon_sha256":"40c47156e35536aaac8097dc25a8d2b9a700e9d6cae01f39db9b667fcef398f1","abstract_canon_sha256":"0d8934607c7804b41f501d831161d17ce90c6f94f56af5e7abd548a4816741b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:54:04.095282Z","signature_b64":"SaggxUBX49IIxnHkZq6kQydgteHD5Igt5VpXB+5hrW2ARoUf+l25ZeG9XU57gCnZPiU2N0FO+HvlFuAaW8zSDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"444272456285f4cbd5a6b2e0c10cf134e101a650adb85cadd74a5d0780e9d7d8","last_reissued_at":"2026-07-05T11:54:04.094835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:54:04.094835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SSRL: Self-Search Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bingning Wang, Bowen Zhou, Che Jiang, Cheng Huang, Gang Chen, Heng Zhou, Kaiyan Zhang, Lei Bai, Li Kang, Ning Ding, Xinwei Long, Xuekai Zhu, Yanxu Chen, Yuchen Fan, Yuchen Zhang, Yu Fu, Yuxin Zuo, Zhizhou He","submitted_at":"2025-08-14T17:46:01Z","abstract_excerpt":"We investigate the potential of large language models (LLMs) to serve as efficient simulators for agentic search tasks in reinforcement learning (RL), thereby reducing dependence on costly interactions with external search engines. To this end, we first quantify the intrinsic search capability of LLMs via structured prompting and repeated sampling, which we term Self-Search. Our results reveal that LLMs exhibit strong scaling behavior with respect to the inference budget, achieving high pass@k on question-answering benchmarks, including the challenging BrowseComp task. Building on these observ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.10874","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.10874/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.10874","created_at":"2026-07-05T11:54:04.094893+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.10874v1","created_at":"2026-07-05T11:54:04.094893+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.10874","created_at":"2026-07-05T11:54:04.094893+00:00"},{"alias_kind":"pith_short_12","alias_value":"IRBHERLCQX2M","created_at":"2026-07-05T11:54:04.094893+00:00"},{"alias_kind":"pith_short_16","alias_value":"IRBHERLCQX2MXVNG","created_at":"2026-07-05T11:54:04.094893+00:00"},{"alias_kind":"pith_short_8","alias_value":"IRBHERLC","created_at":"2026-07-05T11:54:04.094893+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24597","citing_title":"Qwen-AgentWorld: Language World Models for General Agents","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2509.02547","citing_title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","ref_index":292,"is_internal_anchor":false},{"citing_arxiv_id":"2510.00861","citing_title":"Erase to Improve: Erasable Reinforcement Learning for Search-Augmented LLMs","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":132,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IRBHERLCQX2MXVNGWLQMCDHRGT","json":"https://pith.science/pith/IRBHERLCQX2MXVNGWLQMCDHRGT.json","graph_json":"https://pith.science/api/pith-number/IRBHERLCQX2MXVNGWLQMCDHRGT/graph.json","events_json":"https://pith.science/api/pith-number/IRBHERLCQX2MXVNGWLQMCDHRGT/events.json","paper":"https://pith.science/paper/IRBHERLC"},"agent_actions":{"view_html":"https://pith.science/pith/IRBHERLCQX2MXVNGWLQMCDHRGT","download_json":"https://pith.science/pith/IRBHERLCQX2MXVNGWLQMCDHRGT.json","view_paper":"https://pith.science/paper/IRBHERLC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.10874&json=true","fetch_graph":"https://pith.science/api/pith-number/IRBHERLCQX2MXVNGWLQMCDHRGT/graph.json","fetch_events":"https://pith.science/api/pith-number/IRBHERLCQX2MXVNGWLQMCDHRGT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IRBHERLCQX2MXVNGWLQMCDHRGT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IRBHERLCQX2MXVNGWLQMCDHRGT/action/storage_attestation","attest_author":"https://pith.science/pith/IRBHERLCQX2MXVNGWLQMCDHRGT/action/author_attestation","sign_citation":"https://pith.science/pith/IRBHERLCQX2MXVNGWLQMCDHRGT/action/citation_signature","submit_replication":"https://pith.science/pith/IRBHERLCQX2MXVNGWLQMCDHRGT/action/replication_record"}},"created_at":"2026-07-05T11:54:04.094893+00:00","updated_at":"2026-07-05T11:54:04.094893+00:00"}