{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EMBLYIJFZ2LKL53Z3JCSJ7UEBN","short_pith_number":"pith:EMBLYIJF","schema_version":"1.0","canonical_sha256":"2302bc2125ce96a5f779da4524fe840b5671ea994be05a9422dc1fa454a88e93","source":{"kind":"arxiv","id":"2508.09726","version":1},"attestation_state":"computed","paper":{"title":"Sample More to Think Less: Group Filtered Policy Optimization for Concise Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ahmed Awadallah, Dimitris Papailiopoulos, Harkirat Behl, Shivam Garg, Vaishnavi Shrivastava, Vidhisha Balachandran","submitted_at":"2025-08-13T11:43:49Z","abstract_excerpt":"Large language models trained with reinforcement learning with verifiable rewards tend to trade accuracy for length--inflating response lengths to achieve gains in accuracy. While longer answers may be warranted for harder problems, many tokens are merely \"filler\": repetitive, verbose text that makes no real progress. We introduce GFPO (Group Filtered Policy Optimization), which curbs this length explosion by sampling larger groups per problem during training and filtering responses to train on based on two key metrics: (1) response length and (2) token efficiency: reward per token ratio. By s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.09726","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-08-13T11:43:49Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e68f2969217cb6b00b426cee1a5b99ca1212e340f4c64be12dba449af69ab9e6","abstract_canon_sha256":"7fe113052db0a254591ed03f846059bfa4a15c60a96c5c79f3fe64162d5b899a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:20.749738Z","signature_b64":"eYdC53uo1wOKUKkg3WcJYDRKVU8Z6XlACTdU9gLJNibbB+Z3sZmHFlPforDteZvPhPWOI/ez7sLtM52khFIxAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2302bc2125ce96a5f779da4524fe840b5671ea994be05a9422dc1fa454a88e93","last_reissued_at":"2026-07-05T11:53:20.749236Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:20.749236Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sample More to Think Less: Group Filtered Policy Optimization for Concise Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ahmed Awadallah, Dimitris Papailiopoulos, Harkirat Behl, Shivam Garg, Vaishnavi Shrivastava, Vidhisha Balachandran","submitted_at":"2025-08-13T11:43:49Z","abstract_excerpt":"Large language models trained with reinforcement learning with verifiable rewards tend to trade accuracy for length--inflating response lengths to achieve gains in accuracy. While longer answers may be warranted for harder problems, many tokens are merely \"filler\": repetitive, verbose text that makes no real progress. We introduce GFPO (Group Filtered Policy Optimization), which curbs this length explosion by sampling larger groups per problem during training and filtering responses to train on based on two key metrics: (1) response length and (2) token efficiency: reward per token ratio. By s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.09726","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.09726/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.09726","created_at":"2026-07-05T11:53:20.749295+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.09726v1","created_at":"2026-07-05T11:53:20.749295+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.09726","created_at":"2026-07-05T11:53:20.749295+00:00"},{"alias_kind":"pith_short_12","alias_value":"EMBLYIJFZ2LK","created_at":"2026-07-05T11:53:20.749295+00:00"},{"alias_kind":"pith_short_16","alias_value":"EMBLYIJFZ2LKL53Z","created_at":"2026-07-05T11:53:20.749295+00:00"},{"alias_kind":"pith_short_8","alias_value":"EMBLYIJF","created_at":"2026-07-05T11:53:20.749295+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":177,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10768","citing_title":"N-GRPO: Embedding-Level Neighbor Mixing for Enhanced Policy Optimization","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05606","citing_title":"Cross-Epoch Adaptive Rollout Optimization for RL Post-Training","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03234","citing_title":"Right Makes Might: Aligning Verified Hidden States Empowers RL Reasoning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":252,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25183","citing_title":"Knowledge Graph-Driven Expert-Level Reasoning for Neuroscience","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26606","citing_title":"Spend Your Rollouts Where It Counts: Rollout Allocation for Group-Based RL Post-Training","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2509.26522","citing_title":"Entropy After </Think> for reasoning model early exiting","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2601.05242","citing_title":"GDPO: Group reward-Decoupled Normalization Policy Optimization for Multi-reward RL Optimization","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16419","citing_title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","ref_index":157,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02795","citing_title":"Rubrics to Tokens: Bridging Response-level Rubrics and Token-level Rewards in Instruction Following Tasks","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08873","citing_title":"CoDistill-GRPO: A Co-Distillation Recipe for Efficient Group Relative Policy Optimization","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09806","citing_title":"LEAD: Length-Efficient Adaptive and Dynamic Reasoning for Large Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06165","citing_title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","ref_index":284,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06523","citing_title":"On the Implicit Reward Overfitting and the Low-rank Dynamics in RLVR","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09852","citing_title":"MEMENTO: Teaching LLMs to Manage Their Own Context","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02913","citing_title":"Generate, Filter, Control, Replay: A Comprehensive Survey of Rollout Strategies for LLM Reinforcement Learning","ref_index":102,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EMBLYIJFZ2LKL53Z3JCSJ7UEBN","json":"https://pith.science/pith/EMBLYIJFZ2LKL53Z3JCSJ7UEBN.json","graph_json":"https://pith.science/api/pith-number/EMBLYIJFZ2LKL53Z3JCSJ7UEBN/graph.json","events_json":"https://pith.science/api/pith-number/EMBLYIJFZ2LKL53Z3JCSJ7UEBN/events.json","paper":"https://pith.science/paper/EMBLYIJF"},"agent_actions":{"view_html":"https://pith.science/pith/EMBLYIJFZ2LKL53Z3JCSJ7UEBN","download_json":"https://pith.science/pith/EMBLYIJFZ2LKL53Z3JCSJ7UEBN.json","view_paper":"https://pith.science/paper/EMBLYIJF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.09726&json=true","fetch_graph":"https://pith.science/api/pith-number/EMBLYIJFZ2LKL53Z3JCSJ7UEBN/graph.json","fetch_events":"https://pith.science/api/pith-number/EMBLYIJFZ2LKL53Z3JCSJ7UEBN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EMBLYIJFZ2LKL53Z3JCSJ7UEBN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EMBLYIJFZ2LKL53Z3JCSJ7UEBN/action/storage_attestation","attest_author":"https://pith.science/pith/EMBLYIJFZ2LKL53Z3JCSJ7UEBN/action/author_attestation","sign_citation":"https://pith.science/pith/EMBLYIJFZ2LKL53Z3JCSJ7UEBN/action/citation_signature","submit_replication":"https://pith.science/pith/EMBLYIJFZ2LKL53Z3JCSJ7UEBN/action/replication_record"}},"created_at":"2026-07-05T11:53:20.749295+00:00","updated_at":"2026-07-05T11:53:20.749295+00:00"}