{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:OQFWU423F2F3WWLG6K2YE6HP3J","short_pith_number":"pith:OQFWU423","canonical_record":{"source":{"id":"2505.18298","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-23T18:44:46Z","cross_cats_sorted":[],"title_canon_sha256":"4570f8c518a9df66e3391e080ffa90cf0557e6535cdc7d7184fc6d7ef843596c","abstract_canon_sha256":"6991b292e0848e2fad1f74a3aef76d41df6ff51c2ea1f6907c86580bcc323ec6"},"schema_version":"1.0"},"canonical_sha256":"740b6a735b2e8bbb5966f2b58278efda40cae8ba288c8f07c70f6855eec32a67","source":{"kind":"arxiv","id":"2505.18298","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.18298","created_at":"2026-07-05T11:08:44Z"},{"alias_kind":"arxiv_version","alias_value":"2505.18298v1","created_at":"2026-07-05T11:08:44Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18298","created_at":"2026-07-05T11:08:44Z"},{"alias_kind":"pith_short_12","alias_value":"OQFWU423F2F3","created_at":"2026-07-05T11:08:44Z"},{"alias_kind":"pith_short_16","alias_value":"OQFWU423F2F3WWLG","created_at":"2026-07-05T11:08:44Z"},{"alias_kind":"pith_short_8","alias_value":"OQFWU423","created_at":"2026-07-05T11:08:44Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:OQFWU423F2F3WWLG6K2YE6HP3J","target":"record","payload":{"canonical_record":{"source":{"id":"2505.18298","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-23T18:44:46Z","cross_cats_sorted":[],"title_canon_sha256":"4570f8c518a9df66e3391e080ffa90cf0557e6535cdc7d7184fc6d7ef843596c","abstract_canon_sha256":"6991b292e0848e2fad1f74a3aef76d41df6ff51c2ea1f6907c86580bcc323ec6"},"schema_version":"1.0"},"canonical_sha256":"740b6a735b2e8bbb5966f2b58278efda40cae8ba288c8f07c70f6855eec32a67","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:44.354302Z","signature_b64":"T0T5qMrAZfAiAj6jF9wtzqMw3bubvC8XWkEmzBumo/rgVfPLzPloLlVFOxG3pOeePJg1G6PQ2EWT7iQDS5dOCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"740b6a735b2e8bbb5966f2b58278efda40cae8ba288c8f07c70f6855eec32a67","last_reissued_at":"2026-07-05T11:08:44.353850Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:44.353850Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.18298","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:08:44Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"eWgsKgHxHhm3Wj/ZUIAfs7mgdBniwVaL0VULsJ1mNfkaHn3USkQY4yMqiF+R1ErUV7niYp6/kmXaEJtd2YhJBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T15:45:59.761674Z"},"content_sha256":"f94a6ed6fe385a76bd8f1a3593cdad405a9661bce334c8c4791b36eade03cf3b","schema_version":"1.0","event_id":"sha256:f94a6ed6fe385a76bd8f1a3593cdad405a9661bce334c8c4791b36eade03cf3b"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:OQFWU423F2F3WWLG6K2YE6HP3J","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Thinking Fast and Right: Balancing Accuracy and Reasoning Length with Adaptive Rewards","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Claire Cardie, Jinyan Su","submitted_at":"2025-05-23T18:44:46Z","abstract_excerpt":"Large language models (LLMs) have demonstrated strong reasoning abilities in mathematical tasks, often enhanced through reinforcement learning (RL). However, RL-trained models frequently produce unnecessarily long reasoning traces -- even for simple queries -- leading to increased inference costs and latency. While recent approaches attempt to control verbosity by adding length penalties to the reward function, these methods rely on fixed penalty terms that are hard to tune and cannot adapt as the model's reasoning capability evolves, limiting their effectiveness. In this work, we propose an a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18298","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.18298/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:08:44Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"kmgqp4SX3+3+4M6g71+zj2ibTJaRSx/DrJI85/FNlY8sgGY/YfYDWOqQKJJvFVSaRUJ63jxklPiYQyaSjeBIBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T15:45:59.762605Z"},"content_sha256":"ffe7bee3095cc9bcf48397e25c6db9b6ed9a7b87e4d3ed1a34d8d343aed4f514","schema_version":"1.0","event_id":"sha256:ffe7bee3095cc9bcf48397e25c6db9b6ed9a7b87e4d3ed1a34d8d343aed4f514"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J/bundle.json","state_url":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/OQFWU423F2F3WWLG6K2YE6HP3J/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T15:45:59Z","links":{"resolver":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J","bundle":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J/bundle.json","state":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J/state.json","well_known_bundle":"https://pith.science/.well-known/pith/OQFWU423F2F3WWLG6K2YE6HP3J/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:OQFWU423F2F3WWLG6K2YE6HP3J","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6991b292e0848e2fad1f74a3aef76d41df6ff51c2ea1f6907c86580bcc323ec6","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-23T18:44:46Z","title_canon_sha256":"4570f8c518a9df66e3391e080ffa90cf0557e6535cdc7d7184fc6d7ef843596c"},"schema_version":"1.0","source":{"id":"2505.18298","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.18298","created_at":"2026-07-05T11:08:44Z"},{"alias_kind":"arxiv_version","alias_value":"2505.18298v1","created_at":"2026-07-05T11:08:44Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18298","created_at":"2026-07-05T11:08:44Z"},{"alias_kind":"pith_short_12","alias_value":"OQFWU423F2F3","created_at":"2026-07-05T11:08:44Z"},{"alias_kind":"pith_short_16","alias_value":"OQFWU423F2F3WWLG","created_at":"2026-07-05T11:08:44Z"},{"alias_kind":"pith_short_8","alias_value":"OQFWU423","created_at":"2026-07-05T11:08:44Z"}],"graph_snapshots":[{"event_id":"sha256:ffe7bee3095cc9bcf48397e25c6db9b6ed9a7b87e4d3ed1a34d8d343aed4f514","target":"graph","created_at":"2026-07-05T11:08:44Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.18298/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large language models (LLMs) have demonstrated strong reasoning abilities in mathematical tasks, often enhanced through reinforcement learning (RL). However, RL-trained models frequently produce unnecessarily long reasoning traces -- even for simple queries -- leading to increased inference costs and latency. While recent approaches attempt to control verbosity by adding length penalties to the reward function, these methods rely on fixed penalty terms that are hard to tune and cannot adapt as the model's reasoning capability evolves, limiting their effectiveness. In this work, we propose an a","authors_text":"Claire Cardie, Jinyan Su","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-23T18:44:46Z","title":"Thinking Fast and Right: Balancing Accuracy and Reasoning Length with Adaptive Rewards"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18298","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f94a6ed6fe385a76bd8f1a3593cdad405a9661bce334c8c4791b36eade03cf3b","target":"record","created_at":"2026-07-05T11:08:44Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6991b292e0848e2fad1f74a3aef76d41df6ff51c2ea1f6907c86580bcc323ec6","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-23T18:44:46Z","title_canon_sha256":"4570f8c518a9df66e3391e080ffa90cf0557e6535cdc7d7184fc6d7ef843596c"},"schema_version":"1.0","source":{"id":"2505.18298","kind":"arxiv","version":1}},"canonical_sha256":"740b6a735b2e8bbb5966f2b58278efda40cae8ba288c8f07c70f6855eec32a67","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"740b6a735b2e8bbb5966f2b58278efda40cae8ba288c8f07c70f6855eec32a67","first_computed_at":"2026-07-05T11:08:44.353850Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:08:44.353850Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"T0T5qMrAZfAiAj6jF9wtzqMw3bubvC8XWkEmzBumo/rgVfPLzPloLlVFOxG3pOeePJg1G6PQ2EWT7iQDS5dOCw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:08:44.354302Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.18298","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f94a6ed6fe385a76bd8f1a3593cdad405a9661bce334c8c4791b36eade03cf3b","sha256:ffe7bee3095cc9bcf48397e25c6db9b6ed9a7b87e4d3ed1a34d8d343aed4f514"],"state_sha256":"a53a89470ec0e5256751409ac03731e2cf5ba57eb1f5845ae64d929680b8f71e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"rjlNUJA9qDgjNsApP4urv48w1uXLpr6aoXhWHa447Sp4zWuvg2ZDfUARPezWjVal3ql5o7X+yhokH9uxqbCzDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T15:45:59.768337Z","bundle_sha256":"10df82506af6252c151ff802aa26cad0dc0d58e62da21ced51b83c98e758bd44"}}