{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:XE2G3UOSUZJCOL4A3UBGZ4RZ5H","short_pith_number":"pith:XE2G3UOS","canonical_record":{"source":{"id":"2607.10481","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-11T21:22:46Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"3fe1ae3b36fedf41dd383664163624375e1ac5c57139613b670114f109d3228d","abstract_canon_sha256":"f597a629a25f303196c48012094d7f8fd562cef3e095bc9999ee144146cfc7bc"},"schema_version":"1.0"},"canonical_sha256":"b9346dd1d2a652272f80dd026cf239e9e4e591d54ef2045835c840dbb2fbaea0","source":{"kind":"arxiv","id":"2607.10481","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.10481","created_at":"2026-07-14T01:21:17Z"},{"alias_kind":"arxiv_version","alias_value":"2607.10481v1","created_at":"2026-07-14T01:21:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.10481","created_at":"2026-07-14T01:21:17Z"},{"alias_kind":"pith_short_12","alias_value":"XE2G3UOSUZJC","created_at":"2026-07-14T01:21:17Z"},{"alias_kind":"pith_short_16","alias_value":"XE2G3UOSUZJCOL4A","created_at":"2026-07-14T01:21:17Z"},{"alias_kind":"pith_short_8","alias_value":"XE2G3UOS","created_at":"2026-07-14T01:21:17Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:XE2G3UOSUZJCOL4A3UBGZ4RZ5H","target":"record","payload":{"canonical_record":{"source":{"id":"2607.10481","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-11T21:22:46Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"3fe1ae3b36fedf41dd383664163624375e1ac5c57139613b670114f109d3228d","abstract_canon_sha256":"f597a629a25f303196c48012094d7f8fd562cef3e095bc9999ee144146cfc7bc"},"schema_version":"1.0"},"canonical_sha256":"b9346dd1d2a652272f80dd026cf239e9e4e591d54ef2045835c840dbb2fbaea0","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T01:21:17.867699Z","signature_b64":"W7qYcx4GNJSS7cS8N4wWJrbPbqN3Lm8yvQ3hapHMOF0WGaHFDfjfHPcfYmDQH8JFpbZTfMalvzifXsjrp5PXBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9346dd1d2a652272f80dd026cf239e9e4e591d54ef2045835c840dbb2fbaea0","last_reissued_at":"2026-07-14T01:21:17.866837Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T01:21:17.866837Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.10481","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-14T01:21:17Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"AINEVcagRitUXNIJaZpNAVpxRnkI19The/C2dOBXmcdUMM6VzUzSNhW2q3Kn+rkQgsB8znNjXy26oyug8v8kBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T03:34:49.909963Z"},"content_sha256":"4398d32bc155e843db5548a4a881fb0b3cb3496d1ec18592ce474b901ed1b959","schema_version":"1.0","event_id":"sha256:4398d32bc155e843db5548a4a881fb0b3cb3496d1ec18592ce474b901ed1b959"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:XE2G3UOSUZJCOL4A3UBGZ4RZ5H","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"ARMOR: Stabilizing On-Policy LLM RL with Off-Policy Anchor Samples","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chiyu Ma, Guoyin Wang, Jiancan Wu, Jinda Lu, Jingren Zhou, Junkang Wu, Kexin Huang, Shuo Yang, Xiangnan He, Xiang Wang","submitted_at":"2026-07-11T21:22:46Z","abstract_excerpt":"Reinforcement learning (RL) has significantly enhanced the reasoning capabilities of large language models (LLMs), yet the training process remains notoriously fragile. In this work, we investigate a critical source of this instability: over-optimization, where models exploit training heuristics at the expense of generalizable reasoning. While reverse KL regularization is the standard defense against such degradation, our analysis reveals that it is often insufficient in this regime, as it fails to ensure comprehensive coverage of the reference distribution. To address this, we propose ARMOR ("},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.10481","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.10481/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-14T01:21:17Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"QkG9hTMqvO8Aa6vEMpVn87FoT0Jc6AjL40V6Zqe7JNwSXFYxb7IIKPx8/hXiZ16P7d5E23nN/GOQfPP0Y/N6CQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T03:34:49.910935Z"},"content_sha256":"4e4d8f0edeaaa784c3961e020e03eac46b44ddf920717c1fe40cdd8e90111452","schema_version":"1.0","event_id":"sha256:4e4d8f0edeaaa784c3961e020e03eac46b44ddf920717c1fe40cdd8e90111452"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XE2G3UOSUZJCOL4A3UBGZ4RZ5H/bundle.json","state_url":"https://pith.science/pith/XE2G3UOSUZJCOL4A3UBGZ4RZ5H/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XE2G3UOSUZJCOL4A3UBGZ4RZ5H/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T03:34:49Z","links":{"resolver":"https://pith.science/pith/XE2G3UOSUZJCOL4A3UBGZ4RZ5H","bundle":"https://pith.science/pith/XE2G3UOSUZJCOL4A3UBGZ4RZ5H/bundle.json","state":"https://pith.science/pith/XE2G3UOSUZJCOL4A3UBGZ4RZ5H/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XE2G3UOSUZJCOL4A3UBGZ4RZ5H/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:XE2G3UOSUZJCOL4A3UBGZ4RZ5H","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f597a629a25f303196c48012094d7f8fd562cef3e095bc9999ee144146cfc7bc","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-11T21:22:46Z","title_canon_sha256":"3fe1ae3b36fedf41dd383664163624375e1ac5c57139613b670114f109d3228d"},"schema_version":"1.0","source":{"id":"2607.10481","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.10481","created_at":"2026-07-14T01:21:17Z"},{"alias_kind":"arxiv_version","alias_value":"2607.10481v1","created_at":"2026-07-14T01:21:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.10481","created_at":"2026-07-14T01:21:17Z"},{"alias_kind":"pith_short_12","alias_value":"XE2G3UOSUZJC","created_at":"2026-07-14T01:21:17Z"},{"alias_kind":"pith_short_16","alias_value":"XE2G3UOSUZJCOL4A","created_at":"2026-07-14T01:21:17Z"},{"alias_kind":"pith_short_8","alias_value":"XE2G3UOS","created_at":"2026-07-14T01:21:17Z"}],"graph_snapshots":[{"event_id":"sha256:4e4d8f0edeaaa784c3961e020e03eac46b44ddf920717c1fe40cdd8e90111452","target":"graph","created_at":"2026-07-14T01:21:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.10481/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) has significantly enhanced the reasoning capabilities of large language models (LLMs), yet the training process remains notoriously fragile. In this work, we investigate a critical source of this instability: over-optimization, where models exploit training heuristics at the expense of generalizable reasoning. While reverse KL regularization is the standard defense against such degradation, our analysis reveals that it is often insufficient in this regime, as it fails to ensure comprehensive coverage of the reference distribution. To address this, we propose ARMOR (","authors_text":"Chiyu Ma, Guoyin Wang, Jiancan Wu, Jinda Lu, Jingren Zhou, Junkang Wu, Kexin Huang, Shuo Yang, Xiangnan He, Xiang Wang","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-11T21:22:46Z","title":"ARMOR: Stabilizing On-Policy LLM RL with Off-Policy Anchor Samples"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.10481","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4398d32bc155e843db5548a4a881fb0b3cb3496d1ec18592ce474b901ed1b959","target":"record","created_at":"2026-07-14T01:21:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f597a629a25f303196c48012094d7f8fd562cef3e095bc9999ee144146cfc7bc","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-11T21:22:46Z","title_canon_sha256":"3fe1ae3b36fedf41dd383664163624375e1ac5c57139613b670114f109d3228d"},"schema_version":"1.0","source":{"id":"2607.10481","kind":"arxiv","version":1}},"canonical_sha256":"b9346dd1d2a652272f80dd026cf239e9e4e591d54ef2045835c840dbb2fbaea0","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b9346dd1d2a652272f80dd026cf239e9e4e591d54ef2045835c840dbb2fbaea0","first_computed_at":"2026-07-14T01:21:17.866837Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-14T01:21:17.866837Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"W7qYcx4GNJSS7cS8N4wWJrbPbqN3Lm8yvQ3hapHMOF0WGaHFDfjfHPcfYmDQH8JFpbZTfMalvzifXsjrp5PXBw==","signature_status":"signed_v1","signed_at":"2026-07-14T01:21:17.867699Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.10481","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4398d32bc155e843db5548a4a881fb0b3cb3496d1ec18592ce474b901ed1b959","sha256:4e4d8f0edeaaa784c3961e020e03eac46b44ddf920717c1fe40cdd8e90111452"],"state_sha256":"406ecb135a86c122b118d29b20d31648ea855c61f26c646a33e76d94617cb83c"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"p2xlemMCX4/qAoHDIIE3q3GN8yldl6zPjfkIL3ammFS2FrC5DUV2HDetiHn8HTfMEezw7L6ApHLlXNrEPWbICQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T03:34:49.923277Z","bundle_sha256":"3dd69c7f38483c29ff1185cbf7016b4b874eba751e272c7040299f067e51fbd2"}}