{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:6P23HH4CUMFYNPIHS53UEMF6OF","short_pith_number":"pith:6P23HH4C","canonical_record":{"source":{"id":"2506.01096","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-01T17:43:54Z","cross_cats_sorted":[],"title_canon_sha256":"3fdaf1e9001d2294bbe58f60a4318093e4e8696f819825dacc3f5184ed1134fa","abstract_canon_sha256":"5c4925f65062c9ebf7642b1ddd6a9dd7f33586f2105c708287bb7dd5b33910cf"},"schema_version":"1.0"},"canonical_sha256":"f3f5b39f82a30b86bd0797774230be7145ae7b43d42affc5930cb36852e23079","source":{"kind":"arxiv","id":"2506.01096","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.01096","created_at":"2026-07-05T11:50:37Z"},{"alias_kind":"arxiv_version","alias_value":"2506.01096v2","created_at":"2026-07-05T11:50:37Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.01096","created_at":"2026-07-05T11:50:37Z"},{"alias_kind":"pith_short_12","alias_value":"6P23HH4CUMFY","created_at":"2026-07-05T11:50:37Z"},{"alias_kind":"pith_short_16","alias_value":"6P23HH4CUMFYNPIH","created_at":"2026-07-05T11:50:37Z"},{"alias_kind":"pith_short_8","alias_value":"6P23HH4C","created_at":"2026-07-05T11:50:37Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:6P23HH4CUMFYNPIHS53UEMF6OF","target":"record","payload":{"canonical_record":{"source":{"id":"2506.01096","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-01T17:43:54Z","cross_cats_sorted":[],"title_canon_sha256":"3fdaf1e9001d2294bbe58f60a4318093e4e8696f819825dacc3f5184ed1134fa","abstract_canon_sha256":"5c4925f65062c9ebf7642b1ddd6a9dd7f33586f2105c708287bb7dd5b33910cf"},"schema_version":"1.0"},"canonical_sha256":"f3f5b39f82a30b86bd0797774230be7145ae7b43d42affc5930cb36852e23079","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:50:37.358034Z","signature_b64":"HJ/aI5X8pmHqowm7AWjHF4Hi7luW9N87NTBdjtRmcr223mJRgeNMnSIdIwhOKgjTIk49umoGv4F5PCS9SAy+Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3f5b39f82a30b86bd0797774230be7145ae7b43d42affc5930cb36852e23079","last_reissued_at":"2026-07-05T11:50:37.357525Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:50:37.357525Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.01096","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:50:37Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"aadEiyvpFUrvkPHejEZMZh01+j3HRl2kWFKX0BiypQBdSAVZAT6B1mR27fyfnKQwBq7S96sO4taYjqVAySZoAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T22:28:52.928327Z"},"content_sha256":"bfe8dc050cf75ab487f855571a225d894dc7613e6ebd4ee3e68613e6b392790c","schema_version":"1.0","event_id":"sha256:bfe8dc050cf75ab487f855571a225d894dc7613e6ebd4ee3e68613e6b392790c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:6P23HH4CUMFYNPIHS53UEMF6OF","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"SuperRL: Reinforcement Learning with Supervision to Boost Language Model Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dongmei Zhang, Haoyu Dong, Lang Cao, Mengyu Zhou, Shi Han, Shuocheng Li, Xiaojun Ma, Yihao Liu, Yuhang Xie","submitted_at":"2025-06-01T17:43:54Z","abstract_excerpt":"Large language models are increasingly used for complex reasoning tasks where high-quality offline data such as expert-annotated solutions and distilled reasoning traces are often available. However, in environments with sparse rewards, reinforcement learning struggles to sample successful trajectories, leading to inefficient learning. At the same time, these offline trajectories that represent correct reasoning paths are not utilized by standard on-policy reinforcement learning methods. We introduce SuperRL, a unified training framework that adaptively alternates between RL and SFT. Whenever "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.01096","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.01096/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:50:37Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VqxL2sZldGQ5RtzZdZueqAvVwwTNIF3AOXFXBewDgcJW5aweiVHo1v6YuJtbzJdxGmA8of2C8Txg2mIYDG6lCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T22:28:52.929221Z"},"content_sha256":"96031250f3050a5dfad4ca3a836ed9915a7c3d46f93002edc0e95261651a19f2","schema_version":"1.0","event_id":"sha256:96031250f3050a5dfad4ca3a836ed9915a7c3d46f93002edc0e95261651a19f2"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/6P23HH4CUMFYNPIHS53UEMF6OF/bundle.json","state_url":"https://pith.science/pith/6P23HH4CUMFYNPIHS53UEMF6OF/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/6P23HH4CUMFYNPIHS53UEMF6OF/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T22:28:52Z","links":{"resolver":"https://pith.science/pith/6P23HH4CUMFYNPIHS53UEMF6OF","bundle":"https://pith.science/pith/6P23HH4CUMFYNPIHS53UEMF6OF/bundle.json","state":"https://pith.science/pith/6P23HH4CUMFYNPIHS53UEMF6OF/state.json","well_known_bundle":"https://pith.science/.well-known/pith/6P23HH4CUMFYNPIHS53UEMF6OF/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:6P23HH4CUMFYNPIHS53UEMF6OF","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5c4925f65062c9ebf7642b1ddd6a9dd7f33586f2105c708287bb7dd5b33910cf","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-01T17:43:54Z","title_canon_sha256":"3fdaf1e9001d2294bbe58f60a4318093e4e8696f819825dacc3f5184ed1134fa"},"schema_version":"1.0","source":{"id":"2506.01096","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.01096","created_at":"2026-07-05T11:50:37Z"},{"alias_kind":"arxiv_version","alias_value":"2506.01096v2","created_at":"2026-07-05T11:50:37Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.01096","created_at":"2026-07-05T11:50:37Z"},{"alias_kind":"pith_short_12","alias_value":"6P23HH4CUMFY","created_at":"2026-07-05T11:50:37Z"},{"alias_kind":"pith_short_16","alias_value":"6P23HH4CUMFYNPIH","created_at":"2026-07-05T11:50:37Z"},{"alias_kind":"pith_short_8","alias_value":"6P23HH4C","created_at":"2026-07-05T11:50:37Z"}],"graph_snapshots":[{"event_id":"sha256:96031250f3050a5dfad4ca3a836ed9915a7c3d46f93002edc0e95261651a19f2","target":"graph","created_at":"2026-07-05T11:50:37Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.01096/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large language models are increasingly used for complex reasoning tasks where high-quality offline data such as expert-annotated solutions and distilled reasoning traces are often available. However, in environments with sparse rewards, reinforcement learning struggles to sample successful trajectories, leading to inefficient learning. At the same time, these offline trajectories that represent correct reasoning paths are not utilized by standard on-policy reinforcement learning methods. We introduce SuperRL, a unified training framework that adaptively alternates between RL and SFT. Whenever ","authors_text":"Dongmei Zhang, Haoyu Dong, Lang Cao, Mengyu Zhou, Shi Han, Shuocheng Li, Xiaojun Ma, Yihao Liu, Yuhang Xie","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-01T17:43:54Z","title":"SuperRL: Reinforcement Learning with Supervision to Boost Language Model Reasoning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.01096","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:bfe8dc050cf75ab487f855571a225d894dc7613e6ebd4ee3e68613e6b392790c","target":"record","created_at":"2026-07-05T11:50:37Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5c4925f65062c9ebf7642b1ddd6a9dd7f33586f2105c708287bb7dd5b33910cf","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-01T17:43:54Z","title_canon_sha256":"3fdaf1e9001d2294bbe58f60a4318093e4e8696f819825dacc3f5184ed1134fa"},"schema_version":"1.0","source":{"id":"2506.01096","kind":"arxiv","version":2}},"canonical_sha256":"f3f5b39f82a30b86bd0797774230be7145ae7b43d42affc5930cb36852e23079","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f3f5b39f82a30b86bd0797774230be7145ae7b43d42affc5930cb36852e23079","first_computed_at":"2026-07-05T11:50:37.357525Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:50:37.357525Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"HJ/aI5X8pmHqowm7AWjHF4Hi7luW9N87NTBdjtRmcr223mJRgeNMnSIdIwhOKgjTIk49umoGv4F5PCS9SAy+Dg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:50:37.358034Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.01096","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:bfe8dc050cf75ab487f855571a225d894dc7613e6ebd4ee3e68613e6b392790c","sha256:96031250f3050a5dfad4ca3a836ed9915a7c3d46f93002edc0e95261651a19f2"],"state_sha256":"884a744d0687362e5cfac1c48a7ec2b06069f1bc30666dda0b36de68f363f02d"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"qYJqKwwpRUzbCkg2loy+7LP2iZ1wNpQcakL9RZU5QPcqjK0X1A++2ICsqAZaXystTsdCqmohmzdAKb+g5jy4CQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T22:28:52.935760Z","bundle_sha256":"eca4afd200269ab8dd583e648ec9a699d49795a4dc94be267cf5e3e2f542a585"}}