{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:P4HS375LJX5JVRHKLJLWVSPZPR","short_pith_number":"pith:P4HS375L","schema_version":"1.0","canonical_sha256":"7f0f2dffab4dfa9ac4ea5a576ac9f97c4316c0a7ddd3c23f6dbf958134d48add","source":{"kind":"arxiv","id":"2507.21848","version":1},"attestation_state":"computed","paper":{"title":"EDGE-GRPO: Entropy-Driven GRPO with Guided Error Correction for Advantage Diversity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Lei Huang, Siwei Wen, Wenjun Wu, Xingjian Zhang","submitted_at":"2025-07-29T14:23:58Z","abstract_excerpt":"Large Language Models (LLMs) have made remarkable progress in enhancing step-by-step reasoning through reinforcement learning. However, the Group Relative Policy Optimization (GRPO) algorithm, which relies on sparse reward rules, often encounters the issue of identical rewards within groups, leading to the advantage collapse problem. Existing works typically address this challenge from two perspectives: enforcing model reflection to enhance response diversity, and introducing internal feedback to augment the training signal (advantage). In this work, we begin by analyzing the limitations of mo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.21848","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-29T14:23:58Z","cross_cats_sorted":[],"title_canon_sha256":"e88601ae0756490950f4768c3636c29538e341f0d0ad4f8fc695cfed8c5ab174","abstract_canon_sha256":"b8d164e87e3ff4d13a0216498917c5db8110ddd61cdc23ca607cc60140f17fbb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:45:08.658868Z","signature_b64":"5pqtFhSnoIxgsLxdy1lZNMymQ1BagUHgy71B6swHmQhSWVT7pSc+inUMX1vIrGbd5X61UDF3sMpZ5LU2LOxbCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f0f2dffab4dfa9ac4ea5a576ac9f97c4316c0a7ddd3c23f6dbf958134d48add","last_reissued_at":"2026-07-05T11:45:08.658433Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:45:08.658433Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EDGE-GRPO: Entropy-Driven GRPO with Guided Error Correction for Advantage Diversity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Lei Huang, Siwei Wen, Wenjun Wu, Xingjian Zhang","submitted_at":"2025-07-29T14:23:58Z","abstract_excerpt":"Large Language Models (LLMs) have made remarkable progress in enhancing step-by-step reasoning through reinforcement learning. However, the Group Relative Policy Optimization (GRPO) algorithm, which relies on sparse reward rules, often encounters the issue of identical rewards within groups, leading to the advantage collapse problem. Existing works typically address this challenge from two perspectives: enforcing model reflection to enhance response diversity, and introducing internal feedback to augment the training signal (advantage). In this work, we begin by analyzing the limitations of mo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.21848","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.21848/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.21848","created_at":"2026-07-05T11:45:08.658488+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.21848v1","created_at":"2026-07-05T11:45:08.658488+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.21848","created_at":"2026-07-05T11:45:08.658488+00:00"},{"alias_kind":"pith_short_12","alias_value":"P4HS375LJX5J","created_at":"2026-07-05T11:45:08.658488+00:00"},{"alias_kind":"pith_short_16","alias_value":"P4HS375LJX5JVRHK","created_at":"2026-07-05T11:45:08.658488+00:00"},{"alias_kind":"pith_short_8","alias_value":"P4HS375L","created_at":"2026-07-05T11:45:08.658488+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11634","citing_title":"Architecture-Aware Reinforcement Learning Makes Sliding-Window Attention Competitive in Math Reasoning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21125","citing_title":"Advantage Collapse in Group Relative Policy Optimization: Diagnosis and Mitigation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31575","citing_title":"Which Tokens Matter? Adaptive Token Selection for RLVR with the Relative Surprisal Index","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17333","citing_title":"Leveraging Error Diversity in Group Rollouts for Reinforcement Learning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29812","citing_title":"Consistency as Inductive Bias: Learning Cross-View Invariance for Robust Multimodal Reasoning","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30789","citing_title":"Smaller Models are Natural Explorers for Policy-Level Diversity in GRPO","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2510.08141","citing_title":"SCOPE-RL: Stable and Quantitative Control of Policy Entropy in RL Post-Training","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21125","citing_title":"Advantage Collapse in Group Relative Policy Optimization: Diagnosis and Mitigation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17333","citing_title":"Leveraging Error Diversity in Group Rollouts for Reinforcement Learning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2509.02547","citing_title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2511.11113","citing_title":"VIDEOP2R: Video Understanding from Perception to Reasoning","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23318","citing_title":"Hidden States Know Where Reasoning Diverges: Credit Assignment via Span-Level Wasserstein Distance","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08905","citing_title":"StaRPO: Stability-Augmented Reinforcement Policy Optimization","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P4HS375LJX5JVRHKLJLWVSPZPR","json":"https://pith.science/pith/P4HS375LJX5JVRHKLJLWVSPZPR.json","graph_json":"https://pith.science/api/pith-number/P4HS375LJX5JVRHKLJLWVSPZPR/graph.json","events_json":"https://pith.science/api/pith-number/P4HS375LJX5JVRHKLJLWVSPZPR/events.json","paper":"https://pith.science/paper/P4HS375L"},"agent_actions":{"view_html":"https://pith.science/pith/P4HS375LJX5JVRHKLJLWVSPZPR","download_json":"https://pith.science/pith/P4HS375LJX5JVRHKLJLWVSPZPR.json","view_paper":"https://pith.science/paper/P4HS375L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.21848&json=true","fetch_graph":"https://pith.science/api/pith-number/P4HS375LJX5JVRHKLJLWVSPZPR/graph.json","fetch_events":"https://pith.science/api/pith-number/P4HS375LJX5JVRHKLJLWVSPZPR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P4HS375LJX5JVRHKLJLWVSPZPR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P4HS375LJX5JVRHKLJLWVSPZPR/action/storage_attestation","attest_author":"https://pith.science/pith/P4HS375LJX5JVRHKLJLWVSPZPR/action/author_attestation","sign_citation":"https://pith.science/pith/P4HS375LJX5JVRHKLJLWVSPZPR/action/citation_signature","submit_replication":"https://pith.science/pith/P4HS375LJX5JVRHKLJLWVSPZPR/action/replication_record"}},"created_at":"2026-07-05T11:45:08.658488+00:00","updated_at":"2026-07-05T11:45:08.658488+00:00"}