{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:IGAMO5MHAG25OHTTEIRJ3COGGT","short_pith_number":"pith:IGAMO5MH","canonical_record":{"source":{"id":"2607.09492","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-10T15:06:48Z","cross_cats_sorted":[],"title_canon_sha256":"c986712418425de964fd09d765efa09d741595da69a78bfcd78e80d613e2a7c3","abstract_canon_sha256":"ca7778d85ebd1417bc9c1112e030b23fccc5e6ece9c6ea2213515fc863617382"},"schema_version":"1.0"},"canonical_sha256":"4180c7758701b5d71e7322229d89c634cf319a22576868b640dd2425c26e7246","source":{"kind":"arxiv","id":"2607.09492","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.09492","created_at":"2026-07-13T01:20:13Z"},{"alias_kind":"arxiv_version","alias_value":"2607.09492v1","created_at":"2026-07-13T01:20:13Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.09492","created_at":"2026-07-13T01:20:13Z"},{"alias_kind":"pith_short_12","alias_value":"IGAMO5MHAG25","created_at":"2026-07-13T01:20:13Z"},{"alias_kind":"pith_short_16","alias_value":"IGAMO5MHAG25OHTT","created_at":"2026-07-13T01:20:13Z"},{"alias_kind":"pith_short_8","alias_value":"IGAMO5MH","created_at":"2026-07-13T01:20:13Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:IGAMO5MHAG25OHTTEIRJ3COGGT","target":"record","payload":{"canonical_record":{"source":{"id":"2607.09492","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-10T15:06:48Z","cross_cats_sorted":[],"title_canon_sha256":"c986712418425de964fd09d765efa09d741595da69a78bfcd78e80d613e2a7c3","abstract_canon_sha256":"ca7778d85ebd1417bc9c1112e030b23fccc5e6ece9c6ea2213515fc863617382"},"schema_version":"1.0"},"canonical_sha256":"4180c7758701b5d71e7322229d89c634cf319a22576868b640dd2425c26e7246","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-13T01:20:13.701899Z","signature_b64":"oqJD9fb6nsM0on+Q/Jqc/LVFzVfAa/sfXxb5xNCaeCVvJTQcbeSB+VBZOBBxJYqTUgfwTJbtvxYat5Lyvgy9Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4180c7758701b5d71e7322229d89c634cf319a22576868b640dd2425c26e7246","last_reissued_at":"2026-07-13T01:20:13.700855Z","signature_status":"signed_v1","first_computed_at":"2026-07-13T01:20:13.700855Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.09492","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-13T01:20:13Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"DoaEhJNEFy+z5ajG0dY6E7TR+IFcc5Y9Oe77sjc5GpKdaSjY6ZlOua4/BRYEBhQJJD+VEwrHC9/2k64fMKb2AA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T20:57:50.310377Z"},"content_sha256":"8e5f8bae0551d93ac8c0ab8e2f2099e2d5b4a703d55611aec928fa789c732a62","schema_version":"1.0","event_id":"sha256:8e5f8bae0551d93ac8c0ab8e2f2099e2d5b4a703d55611aec928fa789c732a62"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:IGAMO5MHAG25OHTTEIRJ3COGGT","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Multimodal Reward Hacking in Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Anmeng Zhang, Jiayu Yao, Lingrui Mei, Shenghua Liu, Songsong Wang, Yiwei Wang, Yuyao Ge, Zhe Sun","submitted_at":"2026-07-10T15:06:48Z","abstract_excerpt":"Reinforcement learning (RL) is increasingly used to align multimodal large language models (MLLMs), but higher rewards do not always imply better task performance. This risk is amplified when visual evidence is evaluated by text-only or weakly grounded rewards. We study reward hacking in MLLM RL across safety VQA, chart VQA, and stress-test settings, varying reward design, data ambiguity, model scale (2B-32B), and RL algorithm (GRPO, RLOO, DAPO). We introduce Newly Rewarded Failure Rate (NRFR), which measures failures among samples whose proxy reward improves over the SFT baseline. Outcome-onl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.09492","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.09492/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-13T01:20:13Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"fKmQmZTl0YDatkC8TUUyAZ2thB3H0O6+KvBVsQT+WP0MpQZiWk5P2J+5FNhpmvH1mMm1X6O1q3k1SwWymrRYCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T20:57:50.311167Z"},"content_sha256":"7455517509a7f13c885ac012a4338a0b9b9e4d014997dac16fde65d2a3e1fdfa","schema_version":"1.0","event_id":"sha256:7455517509a7f13c885ac012a4338a0b9b9e4d014997dac16fde65d2a3e1fdfa"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/IGAMO5MHAG25OHTTEIRJ3COGGT/bundle.json","state_url":"https://pith.science/pith/IGAMO5MHAG25OHTTEIRJ3COGGT/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/IGAMO5MHAG25OHTTEIRJ3COGGT/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T20:57:50Z","links":{"resolver":"https://pith.science/pith/IGAMO5MHAG25OHTTEIRJ3COGGT","bundle":"https://pith.science/pith/IGAMO5MHAG25OHTTEIRJ3COGGT/bundle.json","state":"https://pith.science/pith/IGAMO5MHAG25OHTTEIRJ3COGGT/state.json","well_known_bundle":"https://pith.science/.well-known/pith/IGAMO5MHAG25OHTTEIRJ3COGGT/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:IGAMO5MHAG25OHTTEIRJ3COGGT","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ca7778d85ebd1417bc9c1112e030b23fccc5e6ece9c6ea2213515fc863617382","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-10T15:06:48Z","title_canon_sha256":"c986712418425de964fd09d765efa09d741595da69a78bfcd78e80d613e2a7c3"},"schema_version":"1.0","source":{"id":"2607.09492","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.09492","created_at":"2026-07-13T01:20:13Z"},{"alias_kind":"arxiv_version","alias_value":"2607.09492v1","created_at":"2026-07-13T01:20:13Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.09492","created_at":"2026-07-13T01:20:13Z"},{"alias_kind":"pith_short_12","alias_value":"IGAMO5MHAG25","created_at":"2026-07-13T01:20:13Z"},{"alias_kind":"pith_short_16","alias_value":"IGAMO5MHAG25OHTT","created_at":"2026-07-13T01:20:13Z"},{"alias_kind":"pith_short_8","alias_value":"IGAMO5MH","created_at":"2026-07-13T01:20:13Z"}],"graph_snapshots":[{"event_id":"sha256:7455517509a7f13c885ac012a4338a0b9b9e4d014997dac16fde65d2a3e1fdfa","target":"graph","created_at":"2026-07-13T01:20:13Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.09492/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) is increasingly used to align multimodal large language models (MLLMs), but higher rewards do not always imply better task performance. This risk is amplified when visual evidence is evaluated by text-only or weakly grounded rewards. We study reward hacking in MLLM RL across safety VQA, chart VQA, and stress-test settings, varying reward design, data ambiguity, model scale (2B-32B), and RL algorithm (GRPO, RLOO, DAPO). We introduce Newly Rewarded Failure Rate (NRFR), which measures failures among samples whose proxy reward improves over the SFT baseline. Outcome-onl","authors_text":"Anmeng Zhang, Jiayu Yao, Lingrui Mei, Shenghua Liu, Songsong Wang, Yiwei Wang, Yuyao Ge, Zhe Sun","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-10T15:06:48Z","title":"Multimodal Reward Hacking in Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.09492","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:8e5f8bae0551d93ac8c0ab8e2f2099e2d5b4a703d55611aec928fa789c732a62","target":"record","created_at":"2026-07-13T01:20:13Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ca7778d85ebd1417bc9c1112e030b23fccc5e6ece9c6ea2213515fc863617382","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-10T15:06:48Z","title_canon_sha256":"c986712418425de964fd09d765efa09d741595da69a78bfcd78e80d613e2a7c3"},"schema_version":"1.0","source":{"id":"2607.09492","kind":"arxiv","version":1}},"canonical_sha256":"4180c7758701b5d71e7322229d89c634cf319a22576868b640dd2425c26e7246","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4180c7758701b5d71e7322229d89c634cf319a22576868b640dd2425c26e7246","first_computed_at":"2026-07-13T01:20:13.700855Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-13T01:20:13.700855Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"oqJD9fb6nsM0on+Q/Jqc/LVFzVfAa/sfXxb5xNCaeCVvJTQcbeSB+VBZOBBxJYqTUgfwTJbtvxYat5Lyvgy9Ag==","signature_status":"signed_v1","signed_at":"2026-07-13T01:20:13.701899Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.09492","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:8e5f8bae0551d93ac8c0ab8e2f2099e2d5b4a703d55611aec928fa789c732a62","sha256:7455517509a7f13c885ac012a4338a0b9b9e4d014997dac16fde65d2a3e1fdfa"],"state_sha256":"71885e72806f8aa6e0aab18bbe88f3b32fb7c591c9e359ad812ec5d04ed27959"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IwLgwj+V/Wrls7Eja98kVg4d+0SDJxgoLMWWAh2+t+OztYqFvvg3Ta50V6uW8KtE+djtxk0YQnPmCH0coBWwAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T20:57:50.318138Z","bundle_sha256":"a2eef97a1b48ea1f78b4e1c144e0a1ba047fe7185ccd9f72f59b817a12240b3e"}}