{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:D7KPWI3OTACBQW4M5BDXTGXADR","short_pith_number":"pith:D7KPWI3O","schema_version":"1.0","canonical_sha256":"1fd4fb236e9804185b8ce847799ae01c7611d135c13e0c04191798deddd7efa5","source":{"kind":"arxiv","id":"2505.07903","version":1},"attestation_state":"computed","paper":{"title":"SEM: Reinforcement Learning for Search-Efficient Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Shiwen Cui, Weiqiang Wang, Zeyang Sha","submitted_at":"2025-05-12T09:45:40Z","abstract_excerpt":"Recent advancements in Large Language Models(LLMs) have demonstrated their capabilities not only in reasoning but also in invoking external tools, particularly search engines. However, teaching models to discern when to invoke search and when to rely on their internal knowledge remains a significant challenge. Existing reinforcement learning approaches often lead to redundant search behaviors, resulting in inefficiencies and over-cost. In this paper, we propose SEM, a novel post-training reinforcement learning framework that explicitly trains LLMs to optimize search usage. By constructing a ba"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.07903","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-12T09:45:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f8052cc4a6c07db718fd936de2a05be0f91193f9fed808082289408116cca96f","abstract_canon_sha256":"61873fce3a172bfadfbe24e9838f2ddd05478d1248091518b6f34a455c87c70c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:01:59.297102Z","signature_b64":"b1u987fQbRYgaXAM24gjBPotw2+e3Le+zl6dEjRxoXYh9rQXbegyRqbsmUQzJQAiFbXTltTF36NM4xe4ywV9Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1fd4fb236e9804185b8ce847799ae01c7611d135c13e0c04191798deddd7efa5","last_reissued_at":"2026-07-05T11:01:59.296595Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:01:59.296595Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SEM: Reinforcement Learning for Search-Efficient Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Shiwen Cui, Weiqiang Wang, Zeyang Sha","submitted_at":"2025-05-12T09:45:40Z","abstract_excerpt":"Recent advancements in Large Language Models(LLMs) have demonstrated their capabilities not only in reasoning but also in invoking external tools, particularly search engines. However, teaching models to discern when to invoke search and when to rely on their internal knowledge remains a significant challenge. Existing reinforcement learning approaches often lead to redundant search behaviors, resulting in inefficiencies and over-cost. In this paper, we propose SEM, a novel post-training reinforcement learning framework that explicitly trains LLMs to optimize search usage. By constructing a ba"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.07903","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.07903/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.07903","created_at":"2026-07-05T11:01:59.296660+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.07903v1","created_at":"2026-07-05T11:01:59.296660+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.07903","created_at":"2026-07-05T11:01:59.296660+00:00"},{"alias_kind":"pith_short_12","alias_value":"D7KPWI3OTACB","created_at":"2026-07-05T11:01:59.296660+00:00"},{"alias_kind":"pith_short_16","alias_value":"D7KPWI3OTACBQW4M","created_at":"2026-07-05T11:01:59.296660+00:00"},{"alias_kind":"pith_short_8","alias_value":"D7KPWI3O","created_at":"2026-07-05T11:01:59.296660+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.07794","citing_title":"HiPRAG: Hierarchical Process Rewards for Efficient Agentic Retrieval Augmented Generation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2510.22977","citing_title":"The Reasoning Trap: How Enhancing LLM Reasoning Amplifies Tool Hallucination","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D7KPWI3OTACBQW4M5BDXTGXADR","json":"https://pith.science/pith/D7KPWI3OTACBQW4M5BDXTGXADR.json","graph_json":"https://pith.science/api/pith-number/D7KPWI3OTACBQW4M5BDXTGXADR/graph.json","events_json":"https://pith.science/api/pith-number/D7KPWI3OTACBQW4M5BDXTGXADR/events.json","paper":"https://pith.science/paper/D7KPWI3O"},"agent_actions":{"view_html":"https://pith.science/pith/D7KPWI3OTACBQW4M5BDXTGXADR","download_json":"https://pith.science/pith/D7KPWI3OTACBQW4M5BDXTGXADR.json","view_paper":"https://pith.science/paper/D7KPWI3O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.07903&json=true","fetch_graph":"https://pith.science/api/pith-number/D7KPWI3OTACBQW4M5BDXTGXADR/graph.json","fetch_events":"https://pith.science/api/pith-number/D7KPWI3OTACBQW4M5BDXTGXADR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D7KPWI3OTACBQW4M5BDXTGXADR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D7KPWI3OTACBQW4M5BDXTGXADR/action/storage_attestation","attest_author":"https://pith.science/pith/D7KPWI3OTACBQW4M5BDXTGXADR/action/author_attestation","sign_citation":"https://pith.science/pith/D7KPWI3OTACBQW4M5BDXTGXADR/action/citation_signature","submit_replication":"https://pith.science/pith/D7KPWI3OTACBQW4M5BDXTGXADR/action/replication_record"}},"created_at":"2026-07-05T11:01:59.296660+00:00","updated_at":"2026-07-05T11:01:59.296660+00:00"}