{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:2GWQW6DNCH3NVSCZ32X5OPYG6X","short_pith_number":"pith:2GWQW6DN","canonical_record":{"source":{"id":"2604.08477","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-09T17:16:07Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"042144a56442920a5a3a992e56722ed146e6cde347139fb5d1b309a6e2222d0c","abstract_canon_sha256":"b48f6c434d015ab9c3cb1ea13298d5da80cbbd397cb1b2642cf9c656b42ebb08"},"schema_version":"1.0"},"canonical_sha256":"d1ad0b786d11f6dac859deafd73f06f5fb63f710f3ada59bf4149f5d70d8a32a","source":{"kind":"arxiv","id":"2604.08477","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.08477","created_at":"2026-06-05T01:14:38Z"},{"alias_kind":"arxiv_version","alias_value":"2604.08477v2","created_at":"2026-06-05T01:14:38Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.08477","created_at":"2026-06-05T01:14:38Z"},{"alias_kind":"pith_short_12","alias_value":"2GWQW6DNCH3N","created_at":"2026-06-05T01:14:38Z"},{"alias_kind":"pith_short_16","alias_value":"2GWQW6DNCH3NVSCZ","created_at":"2026-06-05T01:14:38Z"},{"alias_kind":"pith_short_8","alias_value":"2GWQW6DN","created_at":"2026-06-05T01:14:38Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:2GWQW6DNCH3NVSCZ32X5OPYG6X","target":"record","payload":{"canonical_record":{"source":{"id":"2604.08477","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-09T17:16:07Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"042144a56442920a5a3a992e56722ed146e6cde347139fb5d1b309a6e2222d0c","abstract_canon_sha256":"b48f6c434d015ab9c3cb1ea13298d5da80cbbd397cb1b2642cf9c656b42ebb08"},"schema_version":"1.0"},"canonical_sha256":"d1ad0b786d11f6dac859deafd73f06f5fb63f710f3ada59bf4149f5d70d8a32a","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-05T01:14:38.137758Z","signature_b64":"HpKib/USXmnvEOS6ofE1ndK3MRbXNH+DXdBDAKKuuimLuQS6w/L+DW963e14+3shoZc3TUyV4ATUDBmVyGivDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1ad0b786d11f6dac859deafd73f06f5fb63f710f3ada59bf4149f5d70d8a32a","last_reissued_at":"2026-06-05T01:14:38.136786Z","signature_status":"signed_v1","first_computed_at":"2026-06-05T01:14:38.136786Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.08477","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-05T01:14:38Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/nPYzDqAfFfqafXO1ubrDRNzvp+Y0KIcBe8mDjxSLIxFodi9xT1pndV9rNh5Mr5qurezzPrRYD4ENOu8Vru0Dg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T01:29:37.696991Z"},"content_sha256":"6436c25b712ed3366c4c1ca27460ee917cc7a2d2e9c5a60653246d2c740d9d30","schema_version":"1.0","event_id":"sha256:6436c25b712ed3366c4c1ca27460ee917cc7a2d2e9c5a60653246d2c740d9d30"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:2GWQW6DNCH3NVSCZ32X5OPYG6X","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"SUPERNOVA: Eliciting General Reasoning in LLMs with Reinforcement Learning on Natural Instructions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Curating expert-annotated instruction data for verifiable rewards extends reinforcement learning to general reasoning tasks in language models.","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Ashima Suvarna, Hritik Bansal, Kendrick Phan, Mehrab Beikzadeh, Saadia Gabriel","submitted_at":"2026-04-09T17:16:07Z","abstract_excerpt":"Reinforcement Learning with Verifiable Rewards (RLVR) has substantially improved reasoning in formal domains such as mathematics and code, but extending these gains beyond STEM remains challenging. Extending RLVR beyond STEM is fundamentally constrained by the lack of high-quality verifiable training data. In this work, we introduce SUPERNOVA, a framework for curating RLVR data from natural instruction datasets, which are a rich source of expert-annotated data but are underexplored for RLVR training. Through 100+ controlled RL experiments, we systematically study how to utilize these dataset f"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"models trained on SUPERNOVA outperform strong baselines (e.g., Qwen3.5) on challenging reasoning benchmarks including BBEH, Zebralogic, and MMLU-Pro. In particular, training on SUPERNOVA yields relative improvements of up to 52.8% on BBEH across model sizes, demonstrating the effectiveness of principled data curation for RLVR.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That instruction-tuning datasets with expert-annotated ground-truth encode rich reasoning patterns that can be systematically adapted into high-quality verifiable rewards for RLVR, and that the 100+ controlled experiments isolate the effects of source task selection and mixing from other training variables.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"SUPERNOVA adapts instruction-tuning data for RLVR and achieves up to 52.8% relative gains on general reasoning benchmarks like BBEH through targeted task selection and mixing.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Curating expert-annotated instruction data for verifiable rewards extends reinforcement learning to general reasoning tasks in language models.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"ceff7b82da8e8799bd024ca462d512697e09207b4bdc9f231afdbeadcb05692d"},"source":{"id":"2604.08477","kind":"arxiv","version":2},"verdict":{"id":"dffd74cb-5908-4cd6-b107-85e628e78f31","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T17:12:37.357342Z","strongest_claim":"models trained on SUPERNOVA outperform strong baselines (e.g., Qwen3.5) on challenging reasoning benchmarks including BBEH, Zebralogic, and MMLU-Pro. In particular, training on SUPERNOVA yields relative improvements of up to 52.8% on BBEH across model sizes, demonstrating the effectiveness of principled data curation for RLVR.","one_line_summary":"SUPERNOVA adapts instruction-tuning data for RLVR and achieves up to 52.8% relative gains on general reasoning benchmarks like BBEH through targeted task selection and mixing.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That instruction-tuning datasets with expert-annotated ground-truth encode rich reasoning patterns that can be systematically adapted into high-quality verifiable rewards for RLVR, and that the 100+ controlled experiments isolate the effects of source task selection and mixing from other training variables.","pith_extraction_headline":"Curating expert-annotated instruction data for verifiable rewards extends reinforcement learning to general reasoning tasks in language models."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.08477/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"dffd74cb-5908-4cd6-b107-85e628e78f31"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-05T01:14:38Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"iYYyQMfC2+VTRnc5LywOKv9o1KMkPsWdf4t3ZFFShWQL6FWkQCj2sIWa12LgYS07kC2x3Fhns/SxdqRlDBVQBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T01:29:37.698155Z"},"content_sha256":"db5d7e11174ae86e39ab071e75e171a439f5bc1b5f596d7bdfd926ff0ab99be1","schema_version":"1.0","event_id":"sha256:db5d7e11174ae86e39ab071e75e171a439f5bc1b5f596d7bdfd926ff0ab99be1"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2GWQW6DNCH3NVSCZ32X5OPYG6X/bundle.json","state_url":"https://pith.science/pith/2GWQW6DNCH3NVSCZ32X5OPYG6X/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2GWQW6DNCH3NVSCZ32X5OPYG6X/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T01:29:37Z","links":{"resolver":"https://pith.science/pith/2GWQW6DNCH3NVSCZ32X5OPYG6X","bundle":"https://pith.science/pith/2GWQW6DNCH3NVSCZ32X5OPYG6X/bundle.json","state":"https://pith.science/pith/2GWQW6DNCH3NVSCZ32X5OPYG6X/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2GWQW6DNCH3NVSCZ32X5OPYG6X/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:2GWQW6DNCH3NVSCZ32X5OPYG6X","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b48f6c434d015ab9c3cb1ea13298d5da80cbbd397cb1b2642cf9c656b42ebb08","cross_cats_sorted":["cs.CL","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-09T17:16:07Z","title_canon_sha256":"042144a56442920a5a3a992e56722ed146e6cde347139fb5d1b309a6e2222d0c"},"schema_version":"1.0","source":{"id":"2604.08477","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.08477","created_at":"2026-06-05T01:14:38Z"},{"alias_kind":"arxiv_version","alias_value":"2604.08477v2","created_at":"2026-06-05T01:14:38Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.08477","created_at":"2026-06-05T01:14:38Z"},{"alias_kind":"pith_short_12","alias_value":"2GWQW6DNCH3N","created_at":"2026-06-05T01:14:38Z"},{"alias_kind":"pith_short_16","alias_value":"2GWQW6DNCH3NVSCZ","created_at":"2026-06-05T01:14:38Z"},{"alias_kind":"pith_short_8","alias_value":"2GWQW6DN","created_at":"2026-06-05T01:14:38Z"}],"graph_snapshots":[{"event_id":"sha256:db5d7e11174ae86e39ab071e75e171a439f5bc1b5f596d7bdfd926ff0ab99be1","target":"graph","created_at":"2026-06-05T01:14:38Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"models trained on SUPERNOVA outperform strong baselines (e.g., Qwen3.5) on challenging reasoning benchmarks including BBEH, Zebralogic, and MMLU-Pro. In particular, training on SUPERNOVA yields relative improvements of up to 52.8% on BBEH across model sizes, demonstrating the effectiveness of principled data curation for RLVR."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That instruction-tuning datasets with expert-annotated ground-truth encode rich reasoning patterns that can be systematically adapted into high-quality verifiable rewards for RLVR, and that the 100+ controlled experiments isolate the effects of source task selection and mixing from other training variables."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"SUPERNOVA adapts instruction-tuning data for RLVR and achieves up to 52.8% relative gains on general reasoning benchmarks like BBEH through targeted task selection and mixing."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Curating expert-annotated instruction data for verifiable rewards extends reinforcement learning to general reasoning tasks in language models."}],"snapshot_sha256":"ceff7b82da8e8799bd024ca462d512697e09207b4bdc9f231afdbeadcb05692d"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2604.08477/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning with Verifiable Rewards (RLVR) has substantially improved reasoning in formal domains such as mathematics and code, but extending these gains beyond STEM remains challenging. Extending RLVR beyond STEM is fundamentally constrained by the lack of high-quality verifiable training data. In this work, we introduce SUPERNOVA, a framework for curating RLVR data from natural instruction datasets, which are a rich source of expert-annotated data but are underexplored for RLVR training. Through 100+ controlled RL experiments, we systematically study how to utilize these dataset f","authors_text":"Ashima Suvarna, Hritik Bansal, Kendrick Phan, Mehrab Beikzadeh, Saadia Gabriel","cross_cats":["cs.CL","cs.LG"],"headline":"Curating expert-annotated instruction data for verifiable rewards extends reinforcement learning to general reasoning tasks in language models.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-09T17:16:07Z","title":"SUPERNOVA: Eliciting General Reasoning in LLMs with Reinforcement Learning on Natural Instructions"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.08477","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-10T17:12:37.357342Z","id":"dffd74cb-5908-4cd6-b107-85e628e78f31","model_set":{"reader":"grok-4.3"},"one_line_summary":"SUPERNOVA adapts instruction-tuning data for RLVR and achieves up to 52.8% relative gains on general reasoning benchmarks like BBEH through targeted task selection and mixing.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Curating expert-annotated instruction data for verifiable rewards extends reinforcement learning to general reasoning tasks in language models.","strongest_claim":"models trained on SUPERNOVA outperform strong baselines (e.g., Qwen3.5) on challenging reasoning benchmarks including BBEH, Zebralogic, and MMLU-Pro. In particular, training on SUPERNOVA yields relative improvements of up to 52.8% on BBEH across model sizes, demonstrating the effectiveness of principled data curation for RLVR.","weakest_assumption":"That instruction-tuning datasets with expert-annotated ground-truth encode rich reasoning patterns that can be systematically adapted into high-quality verifiable rewards for RLVR, and that the 100+ controlled experiments isolate the effects of source task selection and mixing from other training variables."}},"verdict_id":"dffd74cb-5908-4cd6-b107-85e628e78f31"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:6436c25b712ed3366c4c1ca27460ee917cc7a2d2e9c5a60653246d2c740d9d30","target":"record","created_at":"2026-06-05T01:14:38Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b48f6c434d015ab9c3cb1ea13298d5da80cbbd397cb1b2642cf9c656b42ebb08","cross_cats_sorted":["cs.CL","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-09T17:16:07Z","title_canon_sha256":"042144a56442920a5a3a992e56722ed146e6cde347139fb5d1b309a6e2222d0c"},"schema_version":"1.0","source":{"id":"2604.08477","kind":"arxiv","version":2}},"canonical_sha256":"d1ad0b786d11f6dac859deafd73f06f5fb63f710f3ada59bf4149f5d70d8a32a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d1ad0b786d11f6dac859deafd73f06f5fb63f710f3ada59bf4149f5d70d8a32a","first_computed_at":"2026-06-05T01:14:38.136786Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-05T01:14:38.136786Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"HpKib/USXmnvEOS6ofE1ndK3MRbXNH+DXdBDAKKuuimLuQS6w/L+DW963e14+3shoZc3TUyV4ATUDBmVyGivDA==","signature_status":"signed_v1","signed_at":"2026-06-05T01:14:38.137758Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.08477","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:6436c25b712ed3366c4c1ca27460ee917cc7a2d2e9c5a60653246d2c740d9d30","sha256:db5d7e11174ae86e39ab071e75e171a439f5bc1b5f596d7bdfd926ff0ab99be1"],"state_sha256":"db3c2074c2ce4cd7f9453e9433fa16624fa4269f2d83bd7b1129066185993787"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"OUjwK7RFDIwOu6D+hq1bDX3s9oBF6RH1Pol/rVfutlzZfU8qFMMJ1Bn5NBSh0Yfi+LKbs3JCvzdlyJpgbcoZAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T01:29:37.705810Z","bundle_sha256":"8133662ab6fdee6516655a99534798b04443a8d11f7782ed03a47456d488ac0a"}}