{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:N5T4YVRZZSJAK4YCQEWV7UOPYT","short_pith_number":"pith:N5T4YVRZ","canonical_record":{"source":{"id":"2607.16205","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-07T13:16:09Z","cross_cats_sorted":[],"title_canon_sha256":"1822c959296427e037bc53d82afbf706445dd50fd23a2028a4538171c1b8b292","abstract_canon_sha256":"f823a79a26a811afed5f413d6232886b077f2726240b88852948c139e50b86e4"},"schema_version":"1.0"},"canonical_sha256":"6f67cc5639cc92057302812d5fd1cfc4e9b3be882d1629c41751e2413d6274fe","source":{"kind":"arxiv","id":"2607.16205","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.16205","created_at":"2026-07-21T00:20:05Z"},{"alias_kind":"arxiv_version","alias_value":"2607.16205v1","created_at":"2026-07-21T00:20:05Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.16205","created_at":"2026-07-21T00:20:05Z"},{"alias_kind":"pith_short_12","alias_value":"N5T4YVRZZSJA","created_at":"2026-07-21T00:20:05Z"},{"alias_kind":"pith_short_16","alias_value":"N5T4YVRZZSJAK4YC","created_at":"2026-07-21T00:20:05Z"},{"alias_kind":"pith_short_8","alias_value":"N5T4YVRZ","created_at":"2026-07-21T00:20:05Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:N5T4YVRZZSJAK4YCQEWV7UOPYT","target":"record","payload":{"canonical_record":{"source":{"id":"2607.16205","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-07T13:16:09Z","cross_cats_sorted":[],"title_canon_sha256":"1822c959296427e037bc53d82afbf706445dd50fd23a2028a4538171c1b8b292","abstract_canon_sha256":"f823a79a26a811afed5f413d6232886b077f2726240b88852948c139e50b86e4"},"schema_version":"1.0"},"canonical_sha256":"6f67cc5639cc92057302812d5fd1cfc4e9b3be882d1629c41751e2413d6274fe","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-21T00:20:05.666500Z","signature_b64":"ndOwfOrcWIQZBkcR4eNgVf2jItRTbQ9VM0K9z8kdeU5wwDG2PyVpuMV24cKO8PhXY7gaQfoVFrNFwytYAtljAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6f67cc5639cc92057302812d5fd1cfc4e9b3be882d1629c41751e2413d6274fe","last_reissued_at":"2026-07-21T00:20:05.665457Z","signature_status":"signed_v1","first_computed_at":"2026-07-21T00:20:05.665457Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.16205","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-21T00:20:05Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"NwMJfCeGHpk422hbU/GjtFfdHelJFcDhzmjESmPBdrGlXfhc+1HYyrIF17R9EIODvCuGSKDQevibxA6MDzrOBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T22:57:57.592180Z"},"content_sha256":"674c0d1d404d0fd7a9631550ff06b5d7b9c517e2516412e147ad1c7db95331d4","schema_version":"1.0","event_id":"sha256:674c0d1d404d0fd7a9631550ff06b5d7b9c517e2516412e147ad1c7db95331d4"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:N5T4YVRZZSJAK4YCQEWV7UOPYT","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"It Takes 8 Tokens: Weak-to-Strong Off-Policy RL via Auxiliary Branches","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dayu Wang, Jiahui Liang, Jiaye Yang, Jizhou Huang, Liwei Qian, Weikang Li, Xin Pei","submitted_at":"2026-05-07T13:16:09Z","abstract_excerpt":"Reinforcement learning with verifiable rewards has emerged as a standard approach for enhancing reasoning in large language models, which typically optimizes the policy by contrasting multiple self generated rollouts. However, we identify a critical support limited bottleneck in this paradigm: on challenging reasoning tasks, the target model's samples often exhibit semantic redundancy, converging into the same erroneous \"reasoning basins\" that offer negligible reward contrast for policy updates. In this paper, we propose to overcome this limitation through a weak to strong learning paradigm, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.16205","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.16205/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-21T00:20:05Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"GODTFeWEFh/bwqxHoFQncvetbPBsEZnqag+lvJ3mg+52io/cX6PtQA8hLzt8USo1fepTj2CYJJ5sYAgHQvvOAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T22:57:57.592906Z"},"content_sha256":"230f5e0ae82212677876bc3525ef32590499bd14b0cfe5dad3bb516efc603100","schema_version":"1.0","event_id":"sha256:230f5e0ae82212677876bc3525ef32590499bd14b0cfe5dad3bb516efc603100"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/N5T4YVRZZSJAK4YCQEWV7UOPYT/bundle.json","state_url":"https://pith.science/pith/N5T4YVRZZSJAK4YCQEWV7UOPYT/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/N5T4YVRZZSJAK4YCQEWV7UOPYT/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-19T22:57:57Z","links":{"resolver":"https://pith.science/pith/N5T4YVRZZSJAK4YCQEWV7UOPYT","bundle":"https://pith.science/pith/N5T4YVRZZSJAK4YCQEWV7UOPYT/bundle.json","state":"https://pith.science/pith/N5T4YVRZZSJAK4YCQEWV7UOPYT/state.json","well_known_bundle":"https://pith.science/.well-known/pith/N5T4YVRZZSJAK4YCQEWV7UOPYT/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:N5T4YVRZZSJAK4YCQEWV7UOPYT","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f823a79a26a811afed5f413d6232886b077f2726240b88852948c139e50b86e4","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-07T13:16:09Z","title_canon_sha256":"1822c959296427e037bc53d82afbf706445dd50fd23a2028a4538171c1b8b292"},"schema_version":"1.0","source":{"id":"2607.16205","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.16205","created_at":"2026-07-21T00:20:05Z"},{"alias_kind":"arxiv_version","alias_value":"2607.16205v1","created_at":"2026-07-21T00:20:05Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.16205","created_at":"2026-07-21T00:20:05Z"},{"alias_kind":"pith_short_12","alias_value":"N5T4YVRZZSJA","created_at":"2026-07-21T00:20:05Z"},{"alias_kind":"pith_short_16","alias_value":"N5T4YVRZZSJAK4YC","created_at":"2026-07-21T00:20:05Z"},{"alias_kind":"pith_short_8","alias_value":"N5T4YVRZ","created_at":"2026-07-21T00:20:05Z"}],"graph_snapshots":[{"event_id":"sha256:230f5e0ae82212677876bc3525ef32590499bd14b0cfe5dad3bb516efc603100","target":"graph","created_at":"2026-07-21T00:20:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.16205/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning with verifiable rewards has emerged as a standard approach for enhancing reasoning in large language models, which typically optimizes the policy by contrasting multiple self generated rollouts. However, we identify a critical support limited bottleneck in this paradigm: on challenging reasoning tasks, the target model's samples often exhibit semantic redundancy, converging into the same erroneous \"reasoning basins\" that offer negligible reward contrast for policy updates. In this paper, we propose to overcome this limitation through a weak to strong learning paradigm, w","authors_text":"Dayu Wang, Jiahui Liang, Jiaye Yang, Jizhou Huang, Liwei Qian, Weikang Li, Xin Pei","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-07T13:16:09Z","title":"It Takes 8 Tokens: Weak-to-Strong Off-Policy RL via Auxiliary Branches"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.16205","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:674c0d1d404d0fd7a9631550ff06b5d7b9c517e2516412e147ad1c7db95331d4","target":"record","created_at":"2026-07-21T00:20:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f823a79a26a811afed5f413d6232886b077f2726240b88852948c139e50b86e4","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-07T13:16:09Z","title_canon_sha256":"1822c959296427e037bc53d82afbf706445dd50fd23a2028a4538171c1b8b292"},"schema_version":"1.0","source":{"id":"2607.16205","kind":"arxiv","version":1}},"canonical_sha256":"6f67cc5639cc92057302812d5fd1cfc4e9b3be882d1629c41751e2413d6274fe","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6f67cc5639cc92057302812d5fd1cfc4e9b3be882d1629c41751e2413d6274fe","first_computed_at":"2026-07-21T00:20:05.665457Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-21T00:20:05.665457Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ndOwfOrcWIQZBkcR4eNgVf2jItRTbQ9VM0K9z8kdeU5wwDG2PyVpuMV24cKO8PhXY7gaQfoVFrNFwytYAtljAw==","signature_status":"signed_v1","signed_at":"2026-07-21T00:20:05.666500Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.16205","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:674c0d1d404d0fd7a9631550ff06b5d7b9c517e2516412e147ad1c7db95331d4","sha256:230f5e0ae82212677876bc3525ef32590499bd14b0cfe5dad3bb516efc603100"],"state_sha256":"a8d9359444fbc46e5260441a63e0871eac1214219652c5c94153d33ad0c6967e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zrLpFb1htE4jAVebu+cMQ/a7GEOwjqCloSZbfdSprva8Uz2e+4PTnyK7ELcA7sqGDQD9m+UGu5gZS73mcmlqAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-19T22:57:57.597890Z","bundle_sha256":"f31bbdcf9256373a1e4e9d58e381f8209e456b39bdcdf8c0434f0b483220cde6"}}