{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:RBXXRHLIJVXSPY2JIIGXSQOCWM","short_pith_number":"pith:RBXXRHLI","canonical_record":{"source":{"id":"2506.22950","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-28T16:52:29Z","cross_cats_sorted":[],"title_canon_sha256":"747e2546e4728fe827347839f2132a13197252e4c87180e8b20c568b41819116","abstract_canon_sha256":"766a03861e1becd4a529654ca690efff4e6ed43c81549c9c2f6f8c8b746ec3e2"},"schema_version":"1.0"},"canonical_sha256":"886f789d684d6f27e349420d7941c2b3020143a9e81d2bf549fda753d3c15d99","source":{"kind":"arxiv","id":"2506.22950","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.22950","created_at":"2026-07-05T11:28:48Z"},{"alias_kind":"arxiv_version","alias_value":"2506.22950v1","created_at":"2026-07-05T11:28:48Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.22950","created_at":"2026-07-05T11:28:48Z"},{"alias_kind":"pith_short_12","alias_value":"RBXXRHLIJVXS","created_at":"2026-07-05T11:28:48Z"},{"alias_kind":"pith_short_16","alias_value":"RBXXRHLIJVXSPY2J","created_at":"2026-07-05T11:28:48Z"},{"alias_kind":"pith_short_8","alias_value":"RBXXRHLI","created_at":"2026-07-05T11:28:48Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:RBXXRHLIJVXSPY2JIIGXSQOCWM","target":"record","payload":{"canonical_record":{"source":{"id":"2506.22950","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-28T16:52:29Z","cross_cats_sorted":[],"title_canon_sha256":"747e2546e4728fe827347839f2132a13197252e4c87180e8b20c568b41819116","abstract_canon_sha256":"766a03861e1becd4a529654ca690efff4e6ed43c81549c9c2f6f8c8b746ec3e2"},"schema_version":"1.0"},"canonical_sha256":"886f789d684d6f27e349420d7941c2b3020143a9e81d2bf549fda753d3c15d99","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:48.127289Z","signature_b64":"+i1HBvTaCG8LHl18KtDCT4ktykE281PmXHQKufZ6zp6EhgCzNe8fsr1ABpms/qUUnXTct+HW0FN5lsoJgBTYBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"886f789d684d6f27e349420d7941c2b3020143a9e81d2bf549fda753d3c15d99","last_reissued_at":"2026-07-05T11:28:48.126792Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:48.126792Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.22950","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:28:48Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"WILBwsU9PZjObo5LD4m4KmVw81KHWS63u9ERyF6GJPOjTgF1DpNo7fChx9D2QCYuZetE2Ol66xHqtlnb6bDCCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T06:11:51.209321Z"},"content_sha256":"21f501530be4c2f2f66d195da2f8f1e2727ec48d5ad7470ec2f34399b88c66e9","schema_version":"1.0","event_id":"sha256:21f501530be4c2f2f66d195da2f8f1e2727ec48d5ad7470ec2f34399b88c66e9"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:RBXXRHLIJVXSPY2JIIGXSQOCWM","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Infinite Sampling: Efficient and Stable Grouped RL Training for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Di Wang, Huanyi Xie, Liangyu Wang, Mengdi Li, Tianjin Huang, Xinhai Wang","submitted_at":"2025-06-28T16:52:29Z","abstract_excerpt":"Group-based reinforcement learning algorithms such as Group Reward Policy Optimization (GRPO) have proven effective for fine-tuning large language models (LLMs) with human feedback. However, generating and storing multiple responses per prompt incurs substantial memory overhead, especially as the sample group size increases, limiting scalability under constrained hardware.\n  We propose Infinite Sampling, a framework that enables efficient and stable GRPO training by decoupling group size from GPU memory usage. It consists of: (1) micro sampling groups that decompose large groups into memory-fe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.22950","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.22950/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:28:48Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jqBxYcYrowyEOO8t4wxL+TQrY7CjU90XKIr+hoyrI+bWWppgtm6JkxxqNMn3YIyjKnbLVrjfhpFI0aZiH+T2DQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T06:11:51.210299Z"},"content_sha256":"f85712b26be9d851d866c8054d449510476464b6aa6b6e1209b7142abb754471","schema_version":"1.0","event_id":"sha256:f85712b26be9d851d866c8054d449510476464b6aa6b6e1209b7142abb754471"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RBXXRHLIJVXSPY2JIIGXSQOCWM/bundle.json","state_url":"https://pith.science/pith/RBXXRHLIJVXSPY2JIIGXSQOCWM/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RBXXRHLIJVXSPY2JIIGXSQOCWM/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T06:11:51Z","links":{"resolver":"https://pith.science/pith/RBXXRHLIJVXSPY2JIIGXSQOCWM","bundle":"https://pith.science/pith/RBXXRHLIJVXSPY2JIIGXSQOCWM/bundle.json","state":"https://pith.science/pith/RBXXRHLIJVXSPY2JIIGXSQOCWM/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RBXXRHLIJVXSPY2JIIGXSQOCWM/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:RBXXRHLIJVXSPY2JIIGXSQOCWM","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"766a03861e1becd4a529654ca690efff4e6ed43c81549c9c2f6f8c8b746ec3e2","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-28T16:52:29Z","title_canon_sha256":"747e2546e4728fe827347839f2132a13197252e4c87180e8b20c568b41819116"},"schema_version":"1.0","source":{"id":"2506.22950","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.22950","created_at":"2026-07-05T11:28:48Z"},{"alias_kind":"arxiv_version","alias_value":"2506.22950v1","created_at":"2026-07-05T11:28:48Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.22950","created_at":"2026-07-05T11:28:48Z"},{"alias_kind":"pith_short_12","alias_value":"RBXXRHLIJVXS","created_at":"2026-07-05T11:28:48Z"},{"alias_kind":"pith_short_16","alias_value":"RBXXRHLIJVXSPY2J","created_at":"2026-07-05T11:28:48Z"},{"alias_kind":"pith_short_8","alias_value":"RBXXRHLI","created_at":"2026-07-05T11:28:48Z"}],"graph_snapshots":[{"event_id":"sha256:f85712b26be9d851d866c8054d449510476464b6aa6b6e1209b7142abb754471","target":"graph","created_at":"2026-07-05T11:28:48Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.22950/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Group-based reinforcement learning algorithms such as Group Reward Policy Optimization (GRPO) have proven effective for fine-tuning large language models (LLMs) with human feedback. However, generating and storing multiple responses per prompt incurs substantial memory overhead, especially as the sample group size increases, limiting scalability under constrained hardware.\n  We propose Infinite Sampling, a framework that enables efficient and stable GRPO training by decoupling group size from GPU memory usage. It consists of: (1) micro sampling groups that decompose large groups into memory-fe","authors_text":"Di Wang, Huanyi Xie, Liangyu Wang, Mengdi Li, Tianjin Huang, Xinhai Wang","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-28T16:52:29Z","title":"Infinite Sampling: Efficient and Stable Grouped RL Training for Large Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.22950","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:21f501530be4c2f2f66d195da2f8f1e2727ec48d5ad7470ec2f34399b88c66e9","target":"record","created_at":"2026-07-05T11:28:48Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"766a03861e1becd4a529654ca690efff4e6ed43c81549c9c2f6f8c8b746ec3e2","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-28T16:52:29Z","title_canon_sha256":"747e2546e4728fe827347839f2132a13197252e4c87180e8b20c568b41819116"},"schema_version":"1.0","source":{"id":"2506.22950","kind":"arxiv","version":1}},"canonical_sha256":"886f789d684d6f27e349420d7941c2b3020143a9e81d2bf549fda753d3c15d99","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"886f789d684d6f27e349420d7941c2b3020143a9e81d2bf549fda753d3c15d99","first_computed_at":"2026-07-05T11:28:48.126792Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:28:48.126792Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"+i1HBvTaCG8LHl18KtDCT4ktykE281PmXHQKufZ6zp6EhgCzNe8fsr1ABpms/qUUnXTct+HW0FN5lsoJgBTYBA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:28:48.127289Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.22950","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:21f501530be4c2f2f66d195da2f8f1e2727ec48d5ad7470ec2f34399b88c66e9","sha256:f85712b26be9d851d866c8054d449510476464b6aa6b6e1209b7142abb754471"],"state_sha256":"a9e3e39f54585b09d434679ef0833a0de6ea7e4d6b394e37c4fa214ba5602683"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"N2zVRb2gB98lWLADT5bxNoHn7V6Exyik9KingCPv3HUc8Ut20AFTHMtrIyxgUG0RKDJgKJ23HFViaaUl985SAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T06:11:51.215945Z","bundle_sha256":"24fcf72dd4463e4ebfce69e4ab3c01ca1b5054ba28a0f50f95a436c8d2525ae4"}}