{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:I6TKDVFRCFIDKY7GEJQLONBCEW","short_pith_number":"pith:I6TKDVFR","canonical_record":{"source":{"id":"2607.21793","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-23T20:20:56Z","cross_cats_sorted":[],"title_canon_sha256":"6885a6c1c6d18056bd1ed04a1d2dce84ea7ce076c9b18da769bcf0b39fc5c0ec","abstract_canon_sha256":"3c5ad1ddbb5d3001abd9c5a9418dc72319a239ce7461ccd1f563fb7d21e1c0c2"},"schema_version":"1.0"},"canonical_sha256":"47a6a1d4b111503563e62260b7342225880eca1e0877c526517a9e16cd32f94a","source":{"kind":"arxiv","id":"2607.21793","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.21793","created_at":"2026-07-27T00:20:21Z"},{"alias_kind":"arxiv_version","alias_value":"2607.21793v1","created_at":"2026-07-27T00:20:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.21793","created_at":"2026-07-27T00:20:21Z"},{"alias_kind":"pith_short_12","alias_value":"I6TKDVFRCFID","created_at":"2026-07-27T00:20:21Z"},{"alias_kind":"pith_short_16","alias_value":"I6TKDVFRCFIDKY7G","created_at":"2026-07-27T00:20:21Z"},{"alias_kind":"pith_short_8","alias_value":"I6TKDVFR","created_at":"2026-07-27T00:20:21Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:I6TKDVFRCFIDKY7GEJQLONBCEW","target":"record","payload":{"canonical_record":{"source":{"id":"2607.21793","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-23T20:20:56Z","cross_cats_sorted":[],"title_canon_sha256":"6885a6c1c6d18056bd1ed04a1d2dce84ea7ce076c9b18da769bcf0b39fc5c0ec","abstract_canon_sha256":"3c5ad1ddbb5d3001abd9c5a9418dc72319a239ce7461ccd1f563fb7d21e1c0c2"},"schema_version":"1.0"},"canonical_sha256":"47a6a1d4b111503563e62260b7342225880eca1e0877c526517a9e16cd32f94a","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-27T00:20:21.304688Z","signature_b64":"RGdvIKDGp6smat99l/70L3fyn09YI7YWNHe+DK0lfrdLfIqJYy14GkQclXpi42of+XWUr7ZVPojJ97h6rFZUAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"47a6a1d4b111503563e62260b7342225880eca1e0877c526517a9e16cd32f94a","last_reissued_at":"2026-07-27T00:20:21.303841Z","signature_status":"signed_v1","first_computed_at":"2026-07-27T00:20:21.303841Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.21793","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-27T00:20:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"G0BW/Y7yg+Qk0jsM5fm3fxR0hwEX0fHvIxqRojGr7e//6q6BJM6BybKmz4OVNYPhLCbdS6oTRr/uGKKEUxIjCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T09:34:09.145132Z"},"content_sha256":"a537f06fb511e6bdae63cf78b06248d1e14220a654fd4668ea67604b2084b2e5","schema_version":"1.0","event_id":"sha256:a537f06fb511e6bdae63cf78b06248d1e14220a654fd4668ea67604b2084b2e5"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:I6TKDVFRCFIDKY7GEJQLONBCEW","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"QLPO: Quadrant-weighted Sampling for Length-aware Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bin Cui, Siqi Chen, Siwei Chen, Xupeng Miao","submitted_at":"2026-07-23T20:20:56Z","abstract_excerpt":"Recent large reasoning models often develop long chain-of-thought responses during reinforcement learning (RL), resulting in high inference latency and deployment cost. Existing methods for response length control typically rely on explicit length penalties or additional control modules, which require careful tuning and may compromise reasoning quality. We propose Quadrant-weighted Sampling for Length-aware Policy Optimization (QLPO), a simple resampling-based variant of GRPO that introduces implicit length control without modifying the reward function. QLPO first over-generates candidate resp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.21793","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.21793/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-27T00:20:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"yfbzbaS1NtY57505DO+Siaex922H6EQZgHPOpeXrI3ffBiL5rIR1fSh26uWgxl+PxJAT8JfHZ1x+z4jh5LMZBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T09:34:09.145530Z"},"content_sha256":"ff2c2cf3a63f8bf0e77540f9b2643df88c8e1d9e8bd864747693680e6189272d","schema_version":"1.0","event_id":"sha256:ff2c2cf3a63f8bf0e77540f9b2643df88c8e1d9e8bd864747693680e6189272d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/I6TKDVFRCFIDKY7GEJQLONBCEW/bundle.json","state_url":"https://pith.science/pith/I6TKDVFRCFIDKY7GEJQLONBCEW/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/I6TKDVFRCFIDKY7GEJQLONBCEW/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T09:34:09Z","links":{"resolver":"https://pith.science/pith/I6TKDVFRCFIDKY7GEJQLONBCEW","bundle":"https://pith.science/pith/I6TKDVFRCFIDKY7GEJQLONBCEW/bundle.json","state":"https://pith.science/pith/I6TKDVFRCFIDKY7GEJQLONBCEW/state.json","well_known_bundle":"https://pith.science/.well-known/pith/I6TKDVFRCFIDKY7GEJQLONBCEW/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:I6TKDVFRCFIDKY7GEJQLONBCEW","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"3c5ad1ddbb5d3001abd9c5a9418dc72319a239ce7461ccd1f563fb7d21e1c0c2","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-23T20:20:56Z","title_canon_sha256":"6885a6c1c6d18056bd1ed04a1d2dce84ea7ce076c9b18da769bcf0b39fc5c0ec"},"schema_version":"1.0","source":{"id":"2607.21793","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.21793","created_at":"2026-07-27T00:20:21Z"},{"alias_kind":"arxiv_version","alias_value":"2607.21793v1","created_at":"2026-07-27T00:20:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.21793","created_at":"2026-07-27T00:20:21Z"},{"alias_kind":"pith_short_12","alias_value":"I6TKDVFRCFID","created_at":"2026-07-27T00:20:21Z"},{"alias_kind":"pith_short_16","alias_value":"I6TKDVFRCFIDKY7G","created_at":"2026-07-27T00:20:21Z"},{"alias_kind":"pith_short_8","alias_value":"I6TKDVFR","created_at":"2026-07-27T00:20:21Z"}],"graph_snapshots":[{"event_id":"sha256:ff2c2cf3a63f8bf0e77540f9b2643df88c8e1d9e8bd864747693680e6189272d","target":"graph","created_at":"2026-07-27T00:20:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.21793/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recent large reasoning models often develop long chain-of-thought responses during reinforcement learning (RL), resulting in high inference latency and deployment cost. Existing methods for response length control typically rely on explicit length penalties or additional control modules, which require careful tuning and may compromise reasoning quality. We propose Quadrant-weighted Sampling for Length-aware Policy Optimization (QLPO), a simple resampling-based variant of GRPO that introduces implicit length control without modifying the reward function. QLPO first over-generates candidate resp","authors_text":"Bin Cui, Siqi Chen, Siwei Chen, Xupeng Miao","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-23T20:20:56Z","title":"QLPO: Quadrant-weighted Sampling for Length-aware Policy Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.21793","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:a537f06fb511e6bdae63cf78b06248d1e14220a654fd4668ea67604b2084b2e5","target":"record","created_at":"2026-07-27T00:20:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"3c5ad1ddbb5d3001abd9c5a9418dc72319a239ce7461ccd1f563fb7d21e1c0c2","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-23T20:20:56Z","title_canon_sha256":"6885a6c1c6d18056bd1ed04a1d2dce84ea7ce076c9b18da769bcf0b39fc5c0ec"},"schema_version":"1.0","source":{"id":"2607.21793","kind":"arxiv","version":1}},"canonical_sha256":"47a6a1d4b111503563e62260b7342225880eca1e0877c526517a9e16cd32f94a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"47a6a1d4b111503563e62260b7342225880eca1e0877c526517a9e16cd32f94a","first_computed_at":"2026-07-27T00:20:21.303841Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-27T00:20:21.303841Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"RGdvIKDGp6smat99l/70L3fyn09YI7YWNHe+DK0lfrdLfIqJYy14GkQclXpi42of+XWUr7ZVPojJ97h6rFZUAA==","signature_status":"signed_v1","signed_at":"2026-07-27T00:20:21.304688Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.21793","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:a537f06fb511e6bdae63cf78b06248d1e14220a654fd4668ea67604b2084b2e5","sha256:ff2c2cf3a63f8bf0e77540f9b2643df88c8e1d9e8bd864747693680e6189272d"],"state_sha256":"7ecf6f522d9d6b960c83d03b2ff40a48e5c15268653922782b6208a7b764dc8f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CxpESymevwipvN8Vonw3tnyv13K8WF39D+0OYYB3HV3cbzAxH1YcvCi0dH6IWh9wR+9TmbValnTWg010put6CQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T09:34:09.149008Z","bundle_sha256":"7e5d4f98d1cc7099f397fc98b4b4304dc0d3c4065051818b4c2f45a5e2a49ddb"}}