{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VU7T5FVTVJWWYCKREYX5VC2QDJ","short_pith_number":"pith:VU7T5FVT","schema_version":"1.0","canonical_sha256":"ad3f3e96b3aa6d6c0951262fda8b501a761309312ed5cfdc2da4d2fb0da05982","source":{"kind":"arxiv","id":"2508.11016","version":2},"attestation_state":"computed","paper":{"title":"CURE: Critical-Token-Guided Re-Concatenation for Entropy-Collapse Prevention","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jie Wang, Jing Yang, Miao Liu, Minghui Qiu, Ming Zhou, Qingbin Li, Rongkun Xue, Xiaofeng Ji, Yongqi Wang, Zheming Yang, Zhi Li","submitted_at":"2025-08-14T18:40:34Z","abstract_excerpt":"Recent advances in Reinforcement Learning with Verified Reward (RLVR) have driven the emergence of more sophisticated cognitive behaviors in large language models (LLMs), thereby enhancing their reasoning capabilities. However, in prior RLVR pipelines, the repeated use of static initial-state sampling drawn exactly from the dataset distribution during each sampling phase produced overly deterministic, low diversity model behavior, which manifested as rapid entropy collapse and hindered sustained performance gains during prolonged training. To address this issue, we introduce CURE (Critical-tok"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.11016","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-14T18:40:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"81442bee34a2aee14c9ae3f10b4b2fe9aeb81dfd9f34b179bedfa899f7bdcf5a","abstract_canon_sha256":"755db91f79609bff871423c0217ab534665e92592d5719eec5fdfd058913184b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:58:34.976840Z","signature_b64":"+jwYObNDapR/IbtqwtMz+Q1JuGTzk50W+84Pg5YnWgikyS+0I18Y1lrhN/f9jmbKmmaG6B3rkYLqBgvd/z0vDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ad3f3e96b3aa6d6c0951262fda8b501a761309312ed5cfdc2da4d2fb0da05982","last_reissued_at":"2026-07-05T11:58:34.976341Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:58:34.976341Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CURE: Critical-Token-Guided Re-Concatenation for Entropy-Collapse Prevention","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jie Wang, Jing Yang, Miao Liu, Minghui Qiu, Ming Zhou, Qingbin Li, Rongkun Xue, Xiaofeng Ji, Yongqi Wang, Zheming Yang, Zhi Li","submitted_at":"2025-08-14T18:40:34Z","abstract_excerpt":"Recent advances in Reinforcement Learning with Verified Reward (RLVR) have driven the emergence of more sophisticated cognitive behaviors in large language models (LLMs), thereby enhancing their reasoning capabilities. However, in prior RLVR pipelines, the repeated use of static initial-state sampling drawn exactly from the dataset distribution during each sampling phase produced overly deterministic, low diversity model behavior, which manifested as rapid entropy collapse and hindered sustained performance gains during prolonged training. To address this issue, we introduce CURE (Critical-tok"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.11016","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.11016/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.11016","created_at":"2026-07-05T11:58:34.976406+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.11016v2","created_at":"2026-07-05T11:58:34.976406+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.11016","created_at":"2026-07-05T11:58:34.976406+00:00"},{"alias_kind":"pith_short_12","alias_value":"VU7T5FVTVJWW","created_at":"2026-07-05T11:58:34.976406+00:00"},{"alias_kind":"pith_short_16","alias_value":"VU7T5FVTVJWWYCKR","created_at":"2026-07-05T11:58:34.976406+00:00"},{"alias_kind":"pith_short_8","alias_value":"VU7T5FVT","created_at":"2026-07-05T11:58:34.976406+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.10150","citing_title":"Rethinking Entropy Interventions in RLVR: An Entropy Change Perspective","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VU7T5FVTVJWWYCKREYX5VC2QDJ","json":"https://pith.science/pith/VU7T5FVTVJWWYCKREYX5VC2QDJ.json","graph_json":"https://pith.science/api/pith-number/VU7T5FVTVJWWYCKREYX5VC2QDJ/graph.json","events_json":"https://pith.science/api/pith-number/VU7T5FVTVJWWYCKREYX5VC2QDJ/events.json","paper":"https://pith.science/paper/VU7T5FVT"},"agent_actions":{"view_html":"https://pith.science/pith/VU7T5FVTVJWWYCKREYX5VC2QDJ","download_json":"https://pith.science/pith/VU7T5FVTVJWWYCKREYX5VC2QDJ.json","view_paper":"https://pith.science/paper/VU7T5FVT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.11016&json=true","fetch_graph":"https://pith.science/api/pith-number/VU7T5FVTVJWWYCKREYX5VC2QDJ/graph.json","fetch_events":"https://pith.science/api/pith-number/VU7T5FVTVJWWYCKREYX5VC2QDJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VU7T5FVTVJWWYCKREYX5VC2QDJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VU7T5FVTVJWWYCKREYX5VC2QDJ/action/storage_attestation","attest_author":"https://pith.science/pith/VU7T5FVTVJWWYCKREYX5VC2QDJ/action/author_attestation","sign_citation":"https://pith.science/pith/VU7T5FVTVJWWYCKREYX5VC2QDJ/action/citation_signature","submit_replication":"https://pith.science/pith/VU7T5FVTVJWWYCKREYX5VC2QDJ/action/replication_record"}},"created_at":"2026-07-05T11:58:34.976406+00:00","updated_at":"2026-07-05T11:58:34.976406+00:00"}