{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:3PD4QSV2P64X7KHEPXFBAXM24H","short_pith_number":"pith:3PD4QSV2","canonical_record":{"source":{"id":"2505.17218","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-22T18:48:09Z","cross_cats_sorted":[],"title_canon_sha256":"80f7fa0a168650a3ef7628271cc1d5c33b9a4a5e8a42319be3d8829f3b16b424","abstract_canon_sha256":"043f806e687e8f205e7a5e664a22c90323e3489b29eba8d05f8ebbac77833cb9"},"schema_version":"1.0"},"canonical_sha256":"dbc7c84aba7fb97fa8e47dca105d9ae1fa350de29fb5a13d5f78a5c0e50569b9","source":{"kind":"arxiv","id":"2505.17218","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.17218","created_at":"2026-07-05T11:07:49Z"},{"alias_kind":"arxiv_version","alias_value":"2505.17218v1","created_at":"2026-07-05T11:07:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17218","created_at":"2026-07-05T11:07:49Z"},{"alias_kind":"pith_short_12","alias_value":"3PD4QSV2P64X","created_at":"2026-07-05T11:07:49Z"},{"alias_kind":"pith_short_16","alias_value":"3PD4QSV2P64X7KHE","created_at":"2026-07-05T11:07:49Z"},{"alias_kind":"pith_short_8","alias_value":"3PD4QSV2","created_at":"2026-07-05T11:07:49Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:3PD4QSV2P64X7KHEPXFBAXM24H","target":"record","payload":{"canonical_record":{"source":{"id":"2505.17218","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-22T18:48:09Z","cross_cats_sorted":[],"title_canon_sha256":"80f7fa0a168650a3ef7628271cc1d5c33b9a4a5e8a42319be3d8829f3b16b424","abstract_canon_sha256":"043f806e687e8f205e7a5e664a22c90323e3489b29eba8d05f8ebbac77833cb9"},"schema_version":"1.0"},"canonical_sha256":"dbc7c84aba7fb97fa8e47dca105d9ae1fa350de29fb5a13d5f78a5c0e50569b9","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:49.039091Z","signature_b64":"/AHNSguZdRZsV2WZ2GkxMhKKxE0jZMApsbQtB07Ck2qi+asQ0OG4I4opfm7s4MQ6Oswk0pfNnFCwoaBFep3RCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dbc7c84aba7fb97fa8e47dca105d9ae1fa350de29fb5a13d5f78a5c0e50569b9","last_reissued_at":"2026-07-05T11:07:49.038531Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:49.038531Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.17218","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:07:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"s0nLGM2IBho41zsUeuoc9Yx7yrM5f1rirhJVDGh0BiIiVTe45/rl2H0YF86TpsAiAW/wL8NQ3cyF1PpNl/tGCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T19:12:46.742021Z"},"content_sha256":"c10351bb135a76504b0538af4af137efaf3d9598cda50a241333f9f0535c63a5","schema_version":"1.0","event_id":"sha256:c10351bb135a76504b0538af4af137efaf3d9598cda50a241333f9f0535c63a5"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:3PD4QSV2P64X7KHEPXFBAXM24H","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Effective Reinforcement Learning for Reasoning in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Insup Lee, Lianghuan Huang, Osbert Bastani, Sagnik Anupam, Shuo Li","submitted_at":"2025-05-22T18:48:09Z","abstract_excerpt":"Reinforcement learning (RL) has emerged as a promising strategy for improving the reasoning capabilities of language models (LMs) in domains such as mathematics and coding. However, most modern RL algorithms were designed to target robotics applications, which differ significantly from LM reasoning. We analyze RL algorithm design decisions for LM reasoning, for both accuracy and computational efficiency, focusing on relatively small models due to computational constraints. Our findings are: (i) on-policy RL significantly outperforms supervised fine-tuning (SFT), (ii) PPO-based off-policy updat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17218","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17218/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:07:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"6IZnlE2U9qyti1tzMeZ67CFt3bha3VH/kMjKM7UatfNqI+IP6MykYfJAPtnm/23VldQaERQE8Z2FjASngRhUAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T19:12:46.742895Z"},"content_sha256":"23008a22ab3867d64892662a06046204cd1582808bcda3a7b5f63d6c796a6504","schema_version":"1.0","event_id":"sha256:23008a22ab3867d64892662a06046204cd1582808bcda3a7b5f63d6c796a6504"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/3PD4QSV2P64X7KHEPXFBAXM24H/bundle.json","state_url":"https://pith.science/pith/3PD4QSV2P64X7KHEPXFBAXM24H/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/3PD4QSV2P64X7KHEPXFBAXM24H/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-03T19:12:46Z","links":{"resolver":"https://pith.science/pith/3PD4QSV2P64X7KHEPXFBAXM24H","bundle":"https://pith.science/pith/3PD4QSV2P64X7KHEPXFBAXM24H/bundle.json","state":"https://pith.science/pith/3PD4QSV2P64X7KHEPXFBAXM24H/state.json","well_known_bundle":"https://pith.science/.well-known/pith/3PD4QSV2P64X7KHEPXFBAXM24H/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:3PD4QSV2P64X7KHEPXFBAXM24H","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"043f806e687e8f205e7a5e664a22c90323e3489b29eba8d05f8ebbac77833cb9","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-22T18:48:09Z","title_canon_sha256":"80f7fa0a168650a3ef7628271cc1d5c33b9a4a5e8a42319be3d8829f3b16b424"},"schema_version":"1.0","source":{"id":"2505.17218","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.17218","created_at":"2026-07-05T11:07:49Z"},{"alias_kind":"arxiv_version","alias_value":"2505.17218v1","created_at":"2026-07-05T11:07:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17218","created_at":"2026-07-05T11:07:49Z"},{"alias_kind":"pith_short_12","alias_value":"3PD4QSV2P64X","created_at":"2026-07-05T11:07:49Z"},{"alias_kind":"pith_short_16","alias_value":"3PD4QSV2P64X7KHE","created_at":"2026-07-05T11:07:49Z"},{"alias_kind":"pith_short_8","alias_value":"3PD4QSV2","created_at":"2026-07-05T11:07:49Z"}],"graph_snapshots":[{"event_id":"sha256:23008a22ab3867d64892662a06046204cd1582808bcda3a7b5f63d6c796a6504","target":"graph","created_at":"2026-07-05T11:07:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.17218/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) has emerged as a promising strategy for improving the reasoning capabilities of language models (LMs) in domains such as mathematics and coding. However, most modern RL algorithms were designed to target robotics applications, which differ significantly from LM reasoning. We analyze RL algorithm design decisions for LM reasoning, for both accuracy and computational efficiency, focusing on relatively small models due to computational constraints. Our findings are: (i) on-policy RL significantly outperforms supervised fine-tuning (SFT), (ii) PPO-based off-policy updat","authors_text":"Insup Lee, Lianghuan Huang, Osbert Bastani, Sagnik Anupam, Shuo Li","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-22T18:48:09Z","title":"Effective Reinforcement Learning for Reasoning in Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17218","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c10351bb135a76504b0538af4af137efaf3d9598cda50a241333f9f0535c63a5","target":"record","created_at":"2026-07-05T11:07:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"043f806e687e8f205e7a5e664a22c90323e3489b29eba8d05f8ebbac77833cb9","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-22T18:48:09Z","title_canon_sha256":"80f7fa0a168650a3ef7628271cc1d5c33b9a4a5e8a42319be3d8829f3b16b424"},"schema_version":"1.0","source":{"id":"2505.17218","kind":"arxiv","version":1}},"canonical_sha256":"dbc7c84aba7fb97fa8e47dca105d9ae1fa350de29fb5a13d5f78a5c0e50569b9","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"dbc7c84aba7fb97fa8e47dca105d9ae1fa350de29fb5a13d5f78a5c0e50569b9","first_computed_at":"2026-07-05T11:07:49.038531Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:07:49.038531Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"/AHNSguZdRZsV2WZ2GkxMhKKxE0jZMApsbQtB07Ck2qi+asQ0OG4I4opfm7s4MQ6Oswk0pfNnFCwoaBFep3RCg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:07:49.039091Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.17218","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c10351bb135a76504b0538af4af137efaf3d9598cda50a241333f9f0535c63a5","sha256:23008a22ab3867d64892662a06046204cd1582808bcda3a7b5f63d6c796a6504"],"state_sha256":"556a1a68116edcf04c7fffdd417e80d628505051644425aa58430daebaf7d9c1"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"XQYm5wG7X/9CdqQeK+9uWU/Kda4Ec0vm/RM7o7EDhdYQr+JO2ApQ7X7V7m3Hk3njuMDerA43pG8fU0lER7DADw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-03T19:12:46.763700Z","bundle_sha256":"ef95161f9487a98616ea3a964799cdad4105de29b00bff43f3fe8af9d69c8fc0"}}