{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:BEPVCZ7LO2O65WFQO7MZDWZJWQ","short_pith_number":"pith:BEPVCZ7L","canonical_record":{"source":{"id":"2506.21655","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-26T17:57:08Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"1149fb523bcb42c578ed63a3b877e52fe07d2c25e8df0761cd85d4562b73f06e","abstract_canon_sha256":"6eb070a45e21fcf9fe2f7a8dc7815a9ff940b9ab2c08e3c6b9e7abcf3df18e9f"},"schema_version":"1.0"},"canonical_sha256":"091f5167eb769deed8b077d991db29b433a11f44f716353fc2096c02affb7cdf","source":{"kind":"arxiv","id":"2506.21655","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.21655","created_at":"2026-07-05T11:27:58Z"},{"alias_kind":"arxiv_version","alias_value":"2506.21655v1","created_at":"2026-07-05T11:27:58Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21655","created_at":"2026-07-05T11:27:58Z"},{"alias_kind":"pith_short_12","alias_value":"BEPVCZ7LO2O6","created_at":"2026-07-05T11:27:58Z"},{"alias_kind":"pith_short_16","alias_value":"BEPVCZ7LO2O65WFQ","created_at":"2026-07-05T11:27:58Z"},{"alias_kind":"pith_short_8","alias_value":"BEPVCZ7L","created_at":"2026-07-05T11:27:58Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:BEPVCZ7LO2O65WFQO7MZDWZJWQ","target":"record","payload":{"canonical_record":{"source":{"id":"2506.21655","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-26T17:57:08Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"1149fb523bcb42c578ed63a3b877e52fe07d2c25e8df0761cd85d4562b73f06e","abstract_canon_sha256":"6eb070a45e21fcf9fe2f7a8dc7815a9ff940b9ab2c08e3c6b9e7abcf3df18e9f"},"schema_version":"1.0"},"canonical_sha256":"091f5167eb769deed8b077d991db29b433a11f44f716353fc2096c02affb7cdf","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:58.226240Z","signature_b64":"DeF5TVMy2eJsUoBTT1cKpq3NiTHTG/swGyqlvEmmNjgB0ceF5hVPASKQP0GnpwAw8zqsfz2M5sAIXY4CNzqlCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"091f5167eb769deed8b077d991db29b433a11f44f716353fc2096c02affb7cdf","last_reissued_at":"2026-07-05T11:27:58.225839Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:58.225839Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.21655","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:27:58Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ymNWzYTByQyCq+a1yUxM5JAsh+/OsfJ98NyruRs1Pv7uo9iMPW+HJ0UQW2or8Tdqp15ob8dWfHbFx3NRvdGqDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T20:01:01.740662Z"},"content_sha256":"c7c283e0e8ae8b17ab9e3c9f0884db5d610a6245db66fa38d2137609495fd75c","schema_version":"1.0","event_id":"sha256:c7c283e0e8ae8b17ab9e3c9f0884db5d610a6245db66fa38d2137609495fd75c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:BEPVCZ7LO2O65WFQO7MZDWZJWQ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"APO: Enhancing Reasoning Ability of MLLMs via Asymmetric Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Minjie Hong, Tao Jin, Yan Xia, Zehan Wang, Zhou Zhao, Ziang Zhang, Zirun Guo","submitted_at":"2025-06-26T17:57:08Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) are powerful at integrating diverse data, but they often struggle with complex reasoning. While Reinforcement learning (RL) can boost reasoning in LLMs, applying it to MLLMs is tricky. Common issues include a drop in performance on general tasks and the generation of overly detailed or \"overthinking\" reasoning. Our work investigates how the KL penalty and overthinking affect RL training in MLLMs. We propose Asymmetric Policy Optimization (APO) to address these issues, which divides the sampled responses into positive and negative groups. For positive sa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21655","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21655/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:27:58Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"epJuaexE8KzQhhMB/kuFet2eDmpqduL2ZQnWoyxLD2B3f5hYK8rG5ZQVM5BEbyL4xgylbGZKLSgAn2TDwRL7DA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T20:01:01.741625Z"},"content_sha256":"709b40f0ad3c5c0d05b92a47dde61a5a554031600a228ca4665e3016088b95a4","schema_version":"1.0","event_id":"sha256:709b40f0ad3c5c0d05b92a47dde61a5a554031600a228ca4665e3016088b95a4"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/BEPVCZ7LO2O65WFQO7MZDWZJWQ/bundle.json","state_url":"https://pith.science/pith/BEPVCZ7LO2O65WFQO7MZDWZJWQ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/BEPVCZ7LO2O65WFQO7MZDWZJWQ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T20:01:01Z","links":{"resolver":"https://pith.science/pith/BEPVCZ7LO2O65WFQO7MZDWZJWQ","bundle":"https://pith.science/pith/BEPVCZ7LO2O65WFQO7MZDWZJWQ/bundle.json","state":"https://pith.science/pith/BEPVCZ7LO2O65WFQO7MZDWZJWQ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/BEPVCZ7LO2O65WFQO7MZDWZJWQ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:BEPVCZ7LO2O65WFQO7MZDWZJWQ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6eb070a45e21fcf9fe2f7a8dc7815a9ff940b9ab2c08e3c6b9e7abcf3df18e9f","cross_cats_sorted":["cs.AI","cs.CV"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-26T17:57:08Z","title_canon_sha256":"1149fb523bcb42c578ed63a3b877e52fe07d2c25e8df0761cd85d4562b73f06e"},"schema_version":"1.0","source":{"id":"2506.21655","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.21655","created_at":"2026-07-05T11:27:58Z"},{"alias_kind":"arxiv_version","alias_value":"2506.21655v1","created_at":"2026-07-05T11:27:58Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21655","created_at":"2026-07-05T11:27:58Z"},{"alias_kind":"pith_short_12","alias_value":"BEPVCZ7LO2O6","created_at":"2026-07-05T11:27:58Z"},{"alias_kind":"pith_short_16","alias_value":"BEPVCZ7LO2O65WFQ","created_at":"2026-07-05T11:27:58Z"},{"alias_kind":"pith_short_8","alias_value":"BEPVCZ7L","created_at":"2026-07-05T11:27:58Z"}],"graph_snapshots":[{"event_id":"sha256:709b40f0ad3c5c0d05b92a47dde61a5a554031600a228ca4665e3016088b95a4","target":"graph","created_at":"2026-07-05T11:27:58Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.21655/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Multimodal Large Language Models (MLLMs) are powerful at integrating diverse data, but they often struggle with complex reasoning. While Reinforcement learning (RL) can boost reasoning in LLMs, applying it to MLLMs is tricky. Common issues include a drop in performance on general tasks and the generation of overly detailed or \"overthinking\" reasoning. Our work investigates how the KL penalty and overthinking affect RL training in MLLMs. We propose Asymmetric Policy Optimization (APO) to address these issues, which divides the sampled responses into positive and negative groups. For positive sa","authors_text":"Minjie Hong, Tao Jin, Yan Xia, Zehan Wang, Zhou Zhao, Ziang Zhang, Zirun Guo","cross_cats":["cs.AI","cs.CV"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-26T17:57:08Z","title":"APO: Enhancing Reasoning Ability of MLLMs via Asymmetric Policy Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21655","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c7c283e0e8ae8b17ab9e3c9f0884db5d610a6245db66fa38d2137609495fd75c","target":"record","created_at":"2026-07-05T11:27:58Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6eb070a45e21fcf9fe2f7a8dc7815a9ff940b9ab2c08e3c6b9e7abcf3df18e9f","cross_cats_sorted":["cs.AI","cs.CV"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-26T17:57:08Z","title_canon_sha256":"1149fb523bcb42c578ed63a3b877e52fe07d2c25e8df0761cd85d4562b73f06e"},"schema_version":"1.0","source":{"id":"2506.21655","kind":"arxiv","version":1}},"canonical_sha256":"091f5167eb769deed8b077d991db29b433a11f44f716353fc2096c02affb7cdf","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"091f5167eb769deed8b077d991db29b433a11f44f716353fc2096c02affb7cdf","first_computed_at":"2026-07-05T11:27:58.225839Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:27:58.225839Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"DeF5TVMy2eJsUoBTT1cKpq3NiTHTG/swGyqlvEmmNjgB0ceF5hVPASKQP0GnpwAw8zqsfz2M5sAIXY4CNzqlCg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:27:58.226240Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.21655","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c7c283e0e8ae8b17ab9e3c9f0884db5d610a6245db66fa38d2137609495fd75c","sha256:709b40f0ad3c5c0d05b92a47dde61a5a554031600a228ca4665e3016088b95a4"],"state_sha256":"958db87dfd4c823ebcaea2bd53e166cec292c379d4abb313dc17bf43740b830e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"LDGe4OQWKbEE1y5bGZJFaiYxelqi7o8kXDzUus1HUf4pskPgsmn/APEPB+SDEXR5p9HYyZizpn9TpqtMfNCHAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T20:01:01.764846Z","bundle_sha256":"237024b2c19ba24283bd6287039f5039547885b97e0c6594546c2f7b8c74accb"}}