{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:XGB3Q2TAWZTVMHUL4WENWTOHZN","short_pith_number":"pith:XGB3Q2TA","canonical_record":{"source":{"id":"2607.27271","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T11:39:55Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"4b0495aa339e8d64595f6a1932f31bee085c55cf81fea8ce1c2da594c173ce76","abstract_canon_sha256":"ccefd22a2744884a37e7768714813ada20e8afed999c7c7a3ece56468744daad"},"schema_version":"1.0"},"canonical_sha256":"b983b86a60b667561e8be588db4dc7cb7f10ef4dc129ca5e082fa40b1dcb2460","source":{"kind":"arxiv","id":"2607.27271","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.27271","created_at":"2026-07-31T00:10:28Z"},{"alias_kind":"arxiv_version","alias_value":"2607.27271v1","created_at":"2026-07-31T00:10:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.27271","created_at":"2026-07-31T00:10:28Z"},{"alias_kind":"pith_short_12","alias_value":"XGB3Q2TAWZTV","created_at":"2026-07-31T00:10:28Z"},{"alias_kind":"pith_short_16","alias_value":"XGB3Q2TAWZTVMHUL","created_at":"2026-07-31T00:10:28Z"},{"alias_kind":"pith_short_8","alias_value":"XGB3Q2TA","created_at":"2026-07-31T00:10:28Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:XGB3Q2TAWZTVMHUL4WENWTOHZN","target":"record","payload":{"canonical_record":{"source":{"id":"2607.27271","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T11:39:55Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"4b0495aa339e8d64595f6a1932f31bee085c55cf81fea8ce1c2da594c173ce76","abstract_canon_sha256":"ccefd22a2744884a37e7768714813ada20e8afed999c7c7a3ece56468744daad"},"schema_version":"1.0"},"canonical_sha256":"b983b86a60b667561e8be588db4dc7cb7f10ef4dc129ca5e082fa40b1dcb2460","receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b983b86a60b667561e8be588db4dc7cb7f10ef4dc129ca5e082fa40b1dcb2460","last_reissued_at":"2026-07-31T00:10:28.167131Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-31T00:10:28.167131Z"},"source_kind":"arxiv","source_id":"2607.27271","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-31T00:10:28Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"SnL7yJx5cSqt8BIcSsEWQMLDOL1PNoO5dQwDPJ7uAA2J+OUdEqYCCgtqq/dCM1ugsboRp7+dUHBylKWyTOIwDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T09:30:05.594633Z"},"content_sha256":"10e3b64cca604da162246e4aae005ffad056cd89f2a42d6e59435ee85822b832","schema_version":"1.0","event_id":"sha256:10e3b64cca604da162246e4aae005ffad056cd89f2a42d6e59435ee85822b832"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:XGB3Q2TAWZTVMHUL4WENWTOHZN","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"RLPF: Reinforcement Learning from Performance Feedback for Code Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.LG","authors_text":"Changxuan Fan, Hanyu Yang, Haochen Shi, Haoran Li, Haozhe Cui, Huihao Jing, Shaojin Chen, Sirui Zhang, Wenbin Hu, Yangqiu Song, Yuxuan Liu, Ziyi Chen","submitted_at":"2026-07-29T11:39:55Z","abstract_excerpt":"Code models are increasingly trained with execution feedback, but most training signals still stop at correctness. This leaves an important gap for systems code: two programs can pass the same tests while differing greatly in runtime. We study how to train code agents to prefer faster correct implementations, rather than treating efficiency only as an evaluation metric. The key difficulty is that runtime is a fragile reward. It is meaningful only after a program is correct, varies across tasks, and gives little guidance when most sampled programs fail to compile or run. We propose \\textbf{RLPF"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.27271","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.27271/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-31T00:10:28Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"WwhdH+/NIJSAwMYYqeicLiJ0UirsmlyBx60fWjhGcIDmPKlovD2gGDdTjdzc63PA2e1Z1MfTqCJEzZiYXtYmDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T09:30:05.594993Z"},"content_sha256":"67b80430cdf201278563bb7b5a6850d6b061ef7d6e714b5b0dc7dea037587b7c","schema_version":"1.0","event_id":"sha256:67b80430cdf201278563bb7b5a6850d6b061ef7d6e714b5b0dc7dea037587b7c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XGB3Q2TAWZTVMHUL4WENWTOHZN/bundle.json","state_url":"https://pith.science/pith/XGB3Q2TAWZTVMHUL4WENWTOHZN/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XGB3Q2TAWZTVMHUL4WENWTOHZN/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T09:30:05Z","links":{"resolver":"https://pith.science/pith/XGB3Q2TAWZTVMHUL4WENWTOHZN","bundle":"https://pith.science/pith/XGB3Q2TAWZTVMHUL4WENWTOHZN/bundle.json","state":"https://pith.science/pith/XGB3Q2TAWZTVMHUL4WENWTOHZN/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XGB3Q2TAWZTVMHUL4WENWTOHZN/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:XGB3Q2TAWZTVMHUL4WENWTOHZN","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ccefd22a2744884a37e7768714813ada20e8afed999c7c7a3ece56468744daad","cross_cats_sorted":["cs.SE"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T11:39:55Z","title_canon_sha256":"4b0495aa339e8d64595f6a1932f31bee085c55cf81fea8ce1c2da594c173ce76"},"schema_version":"1.0","source":{"id":"2607.27271","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.27271","created_at":"2026-07-31T00:10:28Z"},{"alias_kind":"arxiv_version","alias_value":"2607.27271v1","created_at":"2026-07-31T00:10:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.27271","created_at":"2026-07-31T00:10:28Z"},{"alias_kind":"pith_short_12","alias_value":"XGB3Q2TAWZTV","created_at":"2026-07-31T00:10:28Z"},{"alias_kind":"pith_short_16","alias_value":"XGB3Q2TAWZTVMHUL","created_at":"2026-07-31T00:10:28Z"},{"alias_kind":"pith_short_8","alias_value":"XGB3Q2TA","created_at":"2026-07-31T00:10:28Z"}],"graph_snapshots":[{"event_id":"sha256:67b80430cdf201278563bb7b5a6850d6b061ef7d6e714b5b0dc7dea037587b7c","target":"graph","created_at":"2026-07-31T00:10:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.27271/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Code models are increasingly trained with execution feedback, but most training signals still stop at correctness. This leaves an important gap for systems code: two programs can pass the same tests while differing greatly in runtime. We study how to train code agents to prefer faster correct implementations, rather than treating efficiency only as an evaluation metric. The key difficulty is that runtime is a fragile reward. It is meaningful only after a program is correct, varies across tasks, and gives little guidance when most sampled programs fail to compile or run. We propose \\textbf{RLPF","authors_text":"Changxuan Fan, Hanyu Yang, Haochen Shi, Haoran Li, Haozhe Cui, Huihao Jing, Shaojin Chen, Sirui Zhang, Wenbin Hu, Yangqiu Song, Yuxuan Liu, Ziyi Chen","cross_cats":["cs.SE"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T11:39:55Z","title":"RLPF: Reinforcement Learning from Performance Feedback for Code Generation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.27271","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:10e3b64cca604da162246e4aae005ffad056cd89f2a42d6e59435ee85822b832","target":"record","created_at":"2026-07-31T00:10:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ccefd22a2744884a37e7768714813ada20e8afed999c7c7a3ece56468744daad","cross_cats_sorted":["cs.SE"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T11:39:55Z","title_canon_sha256":"4b0495aa339e8d64595f6a1932f31bee085c55cf81fea8ce1c2da594c173ce76"},"schema_version":"1.0","source":{"id":"2607.27271","kind":"arxiv","version":1}},"canonical_sha256":"b983b86a60b667561e8be588db4dc7cb7f10ef4dc129ca5e082fa40b1dcb2460","receipt":{"builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b983b86a60b667561e8be588db4dc7cb7f10ef4dc129ca5e082fa40b1dcb2460","first_computed_at":"2026-07-31T00:10:28.167131Z","kind":"pith_receipt","last_reissued_at":"2026-07-31T00:10:28.167131Z","receipt_version":"0.3","signature_status":"unsigned_v0"},"source_id":"2607.27271","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:10e3b64cca604da162246e4aae005ffad056cd89f2a42d6e59435ee85822b832","sha256:67b80430cdf201278563bb7b5a6850d6b061ef7d6e714b5b0dc7dea037587b7c"],"state_sha256":"64bcd2e444383483564f429d6b6440e3e67bcd7f4b832899465785deebc547e1"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5c6YzHo0xzMKYEA3VD8uO7nkNVJ3dbJmclV1939wD1+XRoj+VU9ojUoKEROmPyJem/CHFw/lfwgCk5k0RhFKDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T09:30:05.598484Z","bundle_sha256":"240fd14a2cdc446ef33f962fca79ce96ac31d629ae6c65cf82cd88edff695545"}}