{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:T5QTFBVQQQOA33SIJNKJKYRMWC","short_pith_number":"pith:T5QTFBVQ","canonical_record":{"source":{"id":"2506.05256","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-05T17:17:05Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"13b456c64f9140c0ebb1027512273e9b0b025024316031c211c276df1d38b061","abstract_canon_sha256":"8a0ad8769ae18675831b519676478ceb6e5aa0b5f70ed36ded0b44f6f7ec00a0"},"schema_version":"1.0"},"canonical_sha256":"9f613286b0841c0dee484b5495622cb0aed7a22ce610e438f549b853a18ecfa8","source":{"kind":"arxiv","id":"2506.05256","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.05256","created_at":"2026-07-05T11:17:00Z"},{"alias_kind":"arxiv_version","alias_value":"2506.05256v2","created_at":"2026-07-05T11:17:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.05256","created_at":"2026-07-05T11:17:00Z"},{"alias_kind":"pith_short_12","alias_value":"T5QTFBVQQQOA","created_at":"2026-07-05T11:17:00Z"},{"alias_kind":"pith_short_16","alias_value":"T5QTFBVQQQOA33SI","created_at":"2026-07-05T11:17:00Z"},{"alias_kind":"pith_short_8","alias_value":"T5QTFBVQ","created_at":"2026-07-05T11:17:00Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:T5QTFBVQQQOA33SIJNKJKYRMWC","target":"record","payload":{"canonical_record":{"source":{"id":"2506.05256","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-05T17:17:05Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"13b456c64f9140c0ebb1027512273e9b0b025024316031c211c276df1d38b061","abstract_canon_sha256":"8a0ad8769ae18675831b519676478ceb6e5aa0b5f70ed36ded0b44f6f7ec00a0"},"schema_version":"1.0"},"canonical_sha256":"9f613286b0841c0dee484b5495622cb0aed7a22ce610e438f549b853a18ecfa8","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:00.939928Z","signature_b64":"riREJyBc9FBAvmc2X57EXvYnxUJssf4RyogJ7ZXvnUDGwcXcBSSfk4LUC+Wx5fY8SEuYy96dD+UG80MT3+40CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f613286b0841c0dee484b5495622cb0aed7a22ce610e438f549b853a18ecfa8","last_reissued_at":"2026-07-05T11:17:00.939439Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:00.939439Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.05256","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:17:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ABUNDvvOcPVAZK++IGK6CNp8aHv8R8NOfQLJ1yP9r0aaNJ9ja56eWl4vSZEnE3T6FW4uu8TfyCkah2sNGqA8AA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T04:34:53.548060Z"},"content_sha256":"9144c9b13f016d8152279e17168f12f06787c11192ae412b74ab2c00245e89af","schema_version":"1.0","event_id":"sha256:9144c9b13f016d8152279e17168f12f06787c11192ae412b74ab2c00245e89af"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:T5QTFBVQQQOA33SIJNKJKYRMWC","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Just Enough Thinking: Efficient Reasoning with Adaptive Length Penalties Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Chase Blagden, Chelsea Finn, Nathan Lile, Nick Haber, Rafael Rafailov, Sang Truong, Violet Xiang","submitted_at":"2025-06-05T17:17:05Z","abstract_excerpt":"Large reasoning models (LRMs) achieve higher performance on challenging reasoning tasks by generating more tokens at inference time, but this verbosity often wastes computation on easy problems. Existing solutions, including supervised finetuning on shorter traces, user-controlled budgets, or RL with uniform penalties, either require data curation, manual configuration, or treat all problems alike regardless of difficulty. We introduce Adaptive Length Penalty (ALP), a reinforcement learning objective tailoring generation length to per-prompt solve rate. During training, ALP monitors each promp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.05256","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.05256/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:17:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"x7EedplhAkahrHNA1BVXKQgFW7dxoyNWI1TCgG6m+iecwR9cUwt3L00WczQTBuW15hw04xDnynotzhkvWTz0CA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T04:34:53.548608Z"},"content_sha256":"d19d04d3583f2872439e5af48115923aab683bd91f9231ef91471ae94666d617","schema_version":"1.0","event_id":"sha256:d19d04d3583f2872439e5af48115923aab683bd91f9231ef91471ae94666d617"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC/bundle.json","state_url":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/T5QTFBVQQQOA33SIJNKJKYRMWC/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T04:34:53Z","links":{"resolver":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC","bundle":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC/bundle.json","state":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC/state.json","well_known_bundle":"https://pith.science/.well-known/pith/T5QTFBVQQQOA33SIJNKJKYRMWC/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:T5QTFBVQQQOA33SIJNKJKYRMWC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8a0ad8769ae18675831b519676478ceb6e5aa0b5f70ed36ded0b44f6f7ec00a0","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-05T17:17:05Z","title_canon_sha256":"13b456c64f9140c0ebb1027512273e9b0b025024316031c211c276df1d38b061"},"schema_version":"1.0","source":{"id":"2506.05256","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.05256","created_at":"2026-07-05T11:17:00Z"},{"alias_kind":"arxiv_version","alias_value":"2506.05256v2","created_at":"2026-07-05T11:17:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.05256","created_at":"2026-07-05T11:17:00Z"},{"alias_kind":"pith_short_12","alias_value":"T5QTFBVQQQOA","created_at":"2026-07-05T11:17:00Z"},{"alias_kind":"pith_short_16","alias_value":"T5QTFBVQQQOA33SI","created_at":"2026-07-05T11:17:00Z"},{"alias_kind":"pith_short_8","alias_value":"T5QTFBVQ","created_at":"2026-07-05T11:17:00Z"}],"graph_snapshots":[{"event_id":"sha256:d19d04d3583f2872439e5af48115923aab683bd91f9231ef91471ae94666d617","target":"graph","created_at":"2026-07-05T11:17:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.05256/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large reasoning models (LRMs) achieve higher performance on challenging reasoning tasks by generating more tokens at inference time, but this verbosity often wastes computation on easy problems. Existing solutions, including supervised finetuning on shorter traces, user-controlled budgets, or RL with uniform penalties, either require data curation, manual configuration, or treat all problems alike regardless of difficulty. We introduce Adaptive Length Penalty (ALP), a reinforcement learning objective tailoring generation length to per-prompt solve rate. During training, ALP monitors each promp","authors_text":"Chase Blagden, Chelsea Finn, Nathan Lile, Nick Haber, Rafael Rafailov, Sang Truong, Violet Xiang","cross_cats":["cs.LG"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-05T17:17:05Z","title":"Just Enough Thinking: Efficient Reasoning with Adaptive Length Penalties Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.05256","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:9144c9b13f016d8152279e17168f12f06787c11192ae412b74ab2c00245e89af","target":"record","created_at":"2026-07-05T11:17:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8a0ad8769ae18675831b519676478ceb6e5aa0b5f70ed36ded0b44f6f7ec00a0","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-05T17:17:05Z","title_canon_sha256":"13b456c64f9140c0ebb1027512273e9b0b025024316031c211c276df1d38b061"},"schema_version":"1.0","source":{"id":"2506.05256","kind":"arxiv","version":2}},"canonical_sha256":"9f613286b0841c0dee484b5495622cb0aed7a22ce610e438f549b853a18ecfa8","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9f613286b0841c0dee484b5495622cb0aed7a22ce610e438f549b853a18ecfa8","first_computed_at":"2026-07-05T11:17:00.939439Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:17:00.939439Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"riREJyBc9FBAvmc2X57EXvYnxUJssf4RyogJ7ZXvnUDGwcXcBSSfk4LUC+Wx5fY8SEuYy96dD+UG80MT3+40CA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:17:00.939928Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.05256","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:9144c9b13f016d8152279e17168f12f06787c11192ae412b74ab2c00245e89af","sha256:d19d04d3583f2872439e5af48115923aab683bd91f9231ef91471ae94666d617"],"state_sha256":"d1c7114998a275d7a5e5555af8ee971e5cf54c217667cb37a2e50ef701e0fbcf"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jDxk7L0WYcMxOqa417OEC1CSBbz/XFQpj818TN/zVOqBoNR9UHKm4P/lj2gsEfBylzFkIZMds1ZxkQAR4zM7DQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T04:34:53.553310Z","bundle_sha256":"9235591931dee2c1d1e634a2e50ac12dd1fb0dbd292acb57e1b1c0203ba19644"}}