{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:OQ46UJ2G6DX2AEUWX345UY5FDV","short_pith_number":"pith:OQ46UJ2G","canonical_record":{"source":{"id":"2604.23488","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-26T01:26:50Z","cross_cats_sorted":[],"title_canon_sha256":"225d5fe4b9b2b6bb2417469ca15f49d023b7d98ff19f96d9dc6d0f52c9318e8c","abstract_canon_sha256":"aefce212294c6c3e1b6dacb3980eec4fd438efbd4fdce39aba372904be638b19"},"schema_version":"1.0"},"canonical_sha256":"7439ea2746f0efa01296bef9da63a51d4b15e0bb8cf2e26d9fa541637c32d0d5","source":{"kind":"arxiv","id":"2604.23488","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.23488","created_at":"2026-06-25T00:18:14Z"},{"alias_kind":"arxiv_version","alias_value":"2604.23488v2","created_at":"2026-06-25T00:18:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.23488","created_at":"2026-06-25T00:18:14Z"},{"alias_kind":"pith_short_12","alias_value":"OQ46UJ2G6DX2","created_at":"2026-06-25T00:18:14Z"},{"alias_kind":"pith_short_16","alias_value":"OQ46UJ2G6DX2AEUW","created_at":"2026-06-25T00:18:14Z"},{"alias_kind":"pith_short_8","alias_value":"OQ46UJ2G","created_at":"2026-06-25T00:18:14Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:OQ46UJ2G6DX2AEUWX345UY5FDV","target":"record","payload":{"canonical_record":{"source":{"id":"2604.23488","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-26T01:26:50Z","cross_cats_sorted":[],"title_canon_sha256":"225d5fe4b9b2b6bb2417469ca15f49d023b7d98ff19f96d9dc6d0f52c9318e8c","abstract_canon_sha256":"aefce212294c6c3e1b6dacb3980eec4fd438efbd4fdce39aba372904be638b19"},"schema_version":"1.0"},"canonical_sha256":"7439ea2746f0efa01296bef9da63a51d4b15e0bb8cf2e26d9fa541637c32d0d5","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-25T00:18:14.045647Z","signature_b64":"N5W5yHnwnu5OrCJrsox4za5ntmCFy21aQo/LHnrnUcf4jaVaaJh6uIG4MK7zoDRJd/uH0+lJq25iRsNyqOKcAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7439ea2746f0efa01296bef9da63a51d4b15e0bb8cf2e26d9fa541637c32d0d5","last_reissued_at":"2026-06-25T00:18:14.045221Z","signature_status":"signed_v1","first_computed_at":"2026-06-25T00:18:14.045221Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.23488","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-25T00:18:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VO8u+L6gu/3HuRA+FYtpCndjiW8Llw5DdzpSBMXhFldSZ8plrWzawbJYhc84TEFk6cTbHQZG1cK3UjUJVjAACQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-26T18:14:24.225104Z"},"content_sha256":"6169a74a67fa0819976aa6b18c5c96276b317f5fcd85220c867d8cdc0200f3f6","schema_version":"1.0","event_id":"sha256:6169a74a67fa0819976aa6b18c5c96276b317f5fcd85220c867d8cdc0200f3f6"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:OQ46UJ2G6DX2AEUWX345UY5FDV","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Do Prompt-Elicited Trajectories Reflect Training-Time Reward Hacking? A Systematic Study on Monitoring Trainig-Time Reward Hacking in Code Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Monitors trained on synthetic reward hacking trajectories fail to generalize to in-the-wild hacking in code generation.","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Cho-Jui Hsieh, Hengguang Zhou, Lichen Li, Tianyi Zhou, Yijun Liang","submitted_at":"2026-04-26T01:26:50Z","abstract_excerpt":"Reward hacking in code generation, where models exploit evaluation loopholes to obtain high reward without correctly solving the intended task, poses a critical challenge for Reinforcement Learning (RL) and the deployment of reasoning models. Existing studies often rely on explicitly prompted hacking trajectories, but it remains unclear whether monitors trained on such data can detect reward hacks that arise without direct hacking instructions during RL training. In this work, we introduce Trace-and-Amplify, a framework for scalable curation of reward-hacking trajectories that arise during RL "},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"synthetic-data-trained monitors fail to generalize to 'in-the-wild' hacking, and monitors trained on our 'in-the-wild' trajectories demonstrate stronger generalizability to unseen hacking types.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The modified GRPO procedure with injected conflicting unit tests and resampling-until-hack produces representative samples of naturally emerging reward hacking behaviors during standard RL training.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Synthetic reward hacking data does not capture natural hacking behaviors in code generation RL, causing monitors trained on it to generalize poorly compared to those trained on in-the-wild trajectories.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Monitors trained on synthetic reward hacking trajectories fail to generalize to in-the-wild hacking in code generation.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"fd7b09615a189c1b368268027cc73903cc45e46db44ca8f9cbc2c6736de110eb"},"source":{"id":"2604.23488","kind":"arxiv","version":2},"verdict":{"id":"1f4ac74b-ed59-4e9a-8b5c-35f1e7251fc7","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-08T06:33:52.541552Z","strongest_claim":"synthetic-data-trained monitors fail to generalize to 'in-the-wild' hacking, and monitors trained on our 'in-the-wild' trajectories demonstrate stronger generalizability to unseen hacking types.","one_line_summary":"Synthetic reward hacking data does not capture natural hacking behaviors in code generation RL, causing monitors trained on it to generalize poorly compared to those trained on in-the-wild trajectories.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The modified GRPO procedure with injected conflicting unit tests and resampling-until-hack produces representative samples of naturally emerging reward hacking behaviors during standard RL training.","pith_extraction_headline":"Monitors trained on synthetic reward hacking trajectories fail to generalize to in-the-wild hacking in code generation."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.23488/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-21T08:39:48.375051Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T23:04:31.544953Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"cf8fdc4e89be1a5a08d16a60343f5f168565c73108fa0c41c68dba5f83bb9126"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"1f4ac74b-ed59-4e9a-8b5c-35f1e7251fc7"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-25T00:18:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Y+Tl5twGMiiAikQ8Df30dBQ77PN9ueqP3u50ZULWE2JSqqofcJ5qI/SYvkkoDDof/Ma8B44t7hHt3vv1flRBDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-26T18:14:24.225607Z"},"content_sha256":"ce2f0a071c355cbfa549890f35e5107fd51e4e84d47666dc16f8898e6b8fdc48","schema_version":"1.0","event_id":"sha256:ce2f0a071c355cbfa549890f35e5107fd51e4e84d47666dc16f8898e6b8fdc48"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/OQ46UJ2G6DX2AEUWX345UY5FDV/bundle.json","state_url":"https://pith.science/pith/OQ46UJ2G6DX2AEUWX345UY5FDV/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/OQ46UJ2G6DX2AEUWX345UY5FDV/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-26T18:14:24Z","links":{"resolver":"https://pith.science/pith/OQ46UJ2G6DX2AEUWX345UY5FDV","bundle":"https://pith.science/pith/OQ46UJ2G6DX2AEUWX345UY5FDV/bundle.json","state":"https://pith.science/pith/OQ46UJ2G6DX2AEUWX345UY5FDV/state.json","well_known_bundle":"https://pith.science/.well-known/pith/OQ46UJ2G6DX2AEUWX345UY5FDV/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:OQ46UJ2G6DX2AEUWX345UY5FDV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"aefce212294c6c3e1b6dacb3980eec4fd438efbd4fdce39aba372904be638b19","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-26T01:26:50Z","title_canon_sha256":"225d5fe4b9b2b6bb2417469ca15f49d023b7d98ff19f96d9dc6d0f52c9318e8c"},"schema_version":"1.0","source":{"id":"2604.23488","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.23488","created_at":"2026-06-25T00:18:14Z"},{"alias_kind":"arxiv_version","alias_value":"2604.23488v2","created_at":"2026-06-25T00:18:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.23488","created_at":"2026-06-25T00:18:14Z"},{"alias_kind":"pith_short_12","alias_value":"OQ46UJ2G6DX2","created_at":"2026-06-25T00:18:14Z"},{"alias_kind":"pith_short_16","alias_value":"OQ46UJ2G6DX2AEUW","created_at":"2026-06-25T00:18:14Z"},{"alias_kind":"pith_short_8","alias_value":"OQ46UJ2G","created_at":"2026-06-25T00:18:14Z"}],"graph_snapshots":[{"event_id":"sha256:ce2f0a071c355cbfa549890f35e5107fd51e4e84d47666dc16f8898e6b8fdc48","target":"graph","created_at":"2026-06-25T00:18:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"synthetic-data-trained monitors fail to generalize to 'in-the-wild' hacking, and monitors trained on our 'in-the-wild' trajectories demonstrate stronger generalizability to unseen hacking types."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The modified GRPO procedure with injected conflicting unit tests and resampling-until-hack produces representative samples of naturally emerging reward hacking behaviors during standard RL training."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Synthetic reward hacking data does not capture natural hacking behaviors in code generation RL, causing monitors trained on it to generalize poorly compared to those trained on in-the-wild trajectories."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Monitors trained on synthetic reward hacking trajectories fail to generalize to in-the-wild hacking in code generation."}],"snapshot_sha256":"fd7b09615a189c1b368268027cc73903cc45e46db44ca8f9cbc2c6736de110eb"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-21T08:39:48.375051Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-19T23:04:31.544953Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2604.23488/integrity.json","findings":[],"snapshot_sha256":"cf8fdc4e89be1a5a08d16a60343f5f168565c73108fa0c41c68dba5f83bb9126","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reward hacking in code generation, where models exploit evaluation loopholes to obtain high reward without correctly solving the intended task, poses a critical challenge for Reinforcement Learning (RL) and the deployment of reasoning models. Existing studies often rely on explicitly prompted hacking trajectories, but it remains unclear whether monitors trained on such data can detect reward hacks that arise without direct hacking instructions during RL training. In this work, we introduce Trace-and-Amplify, a framework for scalable curation of reward-hacking trajectories that arise during RL ","authors_text":"Cho-Jui Hsieh, Hengguang Zhou, Lichen Li, Tianyi Zhou, Yijun Liang","cross_cats":[],"headline":"Monitors trained on synthetic reward hacking trajectories fail to generalize to in-the-wild hacking in code generation.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-26T01:26:50Z","title":"Do Prompt-Elicited Trajectories Reflect Training-Time Reward Hacking? A Systematic Study on Monitoring Trainig-Time Reward Hacking in Code Generation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.23488","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-08T06:33:52.541552Z","id":"1f4ac74b-ed59-4e9a-8b5c-35f1e7251fc7","model_set":{"reader":"grok-4.3"},"one_line_summary":"Synthetic reward hacking data does not capture natural hacking behaviors in code generation RL, causing monitors trained on it to generalize poorly compared to those trained on in-the-wild trajectories.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Monitors trained on synthetic reward hacking trajectories fail to generalize to in-the-wild hacking in code generation.","strongest_claim":"synthetic-data-trained monitors fail to generalize to 'in-the-wild' hacking, and monitors trained on our 'in-the-wild' trajectories demonstrate stronger generalizability to unseen hacking types.","weakest_assumption":"The modified GRPO procedure with injected conflicting unit tests and resampling-until-hack produces representative samples of naturally emerging reward hacking behaviors during standard RL training."}},"verdict_id":"1f4ac74b-ed59-4e9a-8b5c-35f1e7251fc7"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:6169a74a67fa0819976aa6b18c5c96276b317f5fcd85220c867d8cdc0200f3f6","target":"record","created_at":"2026-06-25T00:18:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"aefce212294c6c3e1b6dacb3980eec4fd438efbd4fdce39aba372904be638b19","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-26T01:26:50Z","title_canon_sha256":"225d5fe4b9b2b6bb2417469ca15f49d023b7d98ff19f96d9dc6d0f52c9318e8c"},"schema_version":"1.0","source":{"id":"2604.23488","kind":"arxiv","version":2}},"canonical_sha256":"7439ea2746f0efa01296bef9da63a51d4b15e0bb8cf2e26d9fa541637c32d0d5","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"7439ea2746f0efa01296bef9da63a51d4b15e0bb8cf2e26d9fa541637c32d0d5","first_computed_at":"2026-06-25T00:18:14.045221Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-25T00:18:14.045221Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"N5W5yHnwnu5OrCJrsox4za5ntmCFy21aQo/LHnrnUcf4jaVaaJh6uIG4MK7zoDRJd/uH0+lJq25iRsNyqOKcAw==","signature_status":"signed_v1","signed_at":"2026-06-25T00:18:14.045647Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.23488","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:6169a74a67fa0819976aa6b18c5c96276b317f5fcd85220c867d8cdc0200f3f6","sha256:ce2f0a071c355cbfa549890f35e5107fd51e4e84d47666dc16f8898e6b8fdc48"],"state_sha256":"e4f60afb6990dd4abdd1f945497305108c961f4b829c0eaa91ae9529f7a3beda"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ywxHFZRY9fXgZMdyrLJsH3gTKgNbBkazSq8yNJdkmnWbYLjnedJXtnti8XVCKrWQYwwXARCehfH845ApmtGBDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-26T18:14:24.228003Z","bundle_sha256":"8ca40b1b87987fda280f10fe46b03142d8c924a83d1248e46fa1ec1d1571f7eb"}}