{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:AP2LEEU2QSIM5A7PWCFVG2LX7H","short_pith_number":"pith:AP2LEEU2","canonical_record":{"source":{"id":"2604.28123","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-04-30T17:12:53Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"27c5ef591869b652c48eab61e41decb472da3f207ba814ae0c9c6f06475994c0","abstract_canon_sha256":"e3c97e22801ea9e20c07c0c9a43cfe0bc7a0a8791f6969bf1a0a9909cf132dc4"},"schema_version":"1.0"},"canonical_sha256":"03f4b2129a8490ce83efb08b536977f9e2dd7015a1df6c67d3e64af7b5dd0c10","source":{"kind":"arxiv","id":"2604.28123","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.28123","created_at":"2026-06-30T02:17:21Z"},{"alias_kind":"arxiv_version","alias_value":"2604.28123v3","created_at":"2026-06-30T02:17:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.28123","created_at":"2026-06-30T02:17:21Z"},{"alias_kind":"pith_short_12","alias_value":"AP2LEEU2QSIM","created_at":"2026-06-30T02:17:21Z"},{"alias_kind":"pith_short_16","alias_value":"AP2LEEU2QSIM5A7P","created_at":"2026-06-30T02:17:21Z"},{"alias_kind":"pith_short_8","alias_value":"AP2LEEU2","created_at":"2026-06-30T02:17:21Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:AP2LEEU2QSIM5A7PWCFVG2LX7H","target":"record","payload":{"canonical_record":{"source":{"id":"2604.28123","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-04-30T17:12:53Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"27c5ef591869b652c48eab61e41decb472da3f207ba814ae0c9c6f06475994c0","abstract_canon_sha256":"e3c97e22801ea9e20c07c0c9a43cfe0bc7a0a8791f6969bf1a0a9909cf132dc4"},"schema_version":"1.0"},"canonical_sha256":"03f4b2129a8490ce83efb08b536977f9e2dd7015a1df6c67d3e64af7b5dd0c10","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-30T02:17:21.861666Z","signature_b64":"LiwWNg30KQNyEN1ejnKrYYnhNITHcgK3s2M5/gClGllNMcMzmkBt9MwbWlfEgzw1feF6n5Ir36GdLgIzusZCAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03f4b2129a8490ce83efb08b536977f9e2dd7015a1df6c67d3e64af7b5dd0c10","last_reissued_at":"2026-06-30T02:17:21.860964Z","signature_status":"signed_v1","first_computed_at":"2026-06-30T02:17:21.860964Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.28123","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-30T02:17:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HbIMR4P5DMgn87X/LweJ02/O7Zl5pE/gDph3Dn4rx7lDcaoEDpm+lAuM2VZoCSSlCxU1NnfiEzbZGLOFyeQYBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T14:44:49.958164Z"},"content_sha256":"3de6b39fa11a9cfff8b82fc8e0aedc09ba2ea3dca63cd92c1d25883cda41cc23","schema_version":"1.0","event_id":"sha256:3de6b39fa11a9cfff8b82fc8e0aedc09ba2ea3dca63cd92c1d25883cda41cc23"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:AP2LEEU2QSIM5A7PWCFVG2LX7H","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Beyond SFT-to-RL: Pre-alignment via Black-Box On-Policy Distillation for Multimodal RL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Inserting a black-box on-policy distillation stage after SFT corrects distributional drift and raises final multimodal RL accuracy.","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Beier Zhu, Chaojun Xiao, Chen Chen, Chengwei Qin, Hehai Lin, Keming Wu, Sudong Wang, Weiquan Huang, Wenxuan Wang, Xiaomin Yu, Yunjian Zhang, Zuhao Yang","submitted_at":"2026-04-30T17:12:53Z","abstract_excerpt":"The standard post-training recipe for large multimodal models (LMMs) applies supervised fine-tuning (SFT) on curated demonstrations followed by reinforcement learning with verifiable rewards (RLVR). However, SFT introduces distributional drift that neither preserves the model's original capabilities nor faithfully matches the supervision distribution. This problem is further amplified in multimodal reasoning, where perception errors and reasoning failures follow distinct drift patterns that compound during subsequent RL. We introduce PRISM, a three-stage pipeline that mitigates this drift by i"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Experiments on Qwen3-VL show that PRISM consistently improves downstream RLVR performance across multiple RL algorithms (GRPO, DAPO, GSPO) and diverse multimodal benchmarks, improving average accuracy by +4.4 and +6.0 points over the SFT-to-RLVR baseline on 4B and 8B, respectively.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The black-box on-policy distillation game with a perception-reasoning MoE discriminator can supply disentangled corrective signals that steer the policy toward the supervision distribution without access to teacher logits or internal model states.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"PRISM adds a distribution-alignment stage using black-box on-policy distillation against a perception-reasoning MoE discriminator, yielding +4.4 and +6.0 average accuracy gains over standard SFT-to-RLVR on Qwen3-VL 4B and 8B models.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Inserting a black-box on-policy distillation stage after SFT corrects distributional drift and raises final multimodal RL accuracy.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"804495cb19d534fe7931bfcca29387fdb98d2c107ee7127f8f848ced1deebe9a"},"source":{"id":"2604.28123","kind":"arxiv","version":3},"verdict":{"id":"e297ab29-1367-4c9d-9392-48e573dc728f","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-07T07:38:02.491032Z","strongest_claim":"Experiments on Qwen3-VL show that PRISM consistently improves downstream RLVR performance across multiple RL algorithms (GRPO, DAPO, GSPO) and diverse multimodal benchmarks, improving average accuracy by +4.4 and +6.0 points over the SFT-to-RLVR baseline on 4B and 8B, respectively.","one_line_summary":"PRISM adds a distribution-alignment stage using black-box on-policy distillation against a perception-reasoning MoE discriminator, yielding +4.4 and +6.0 average accuracy gains over standard SFT-to-RLVR on Qwen3-VL 4B and 8B models.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The black-box on-policy distillation game with a perception-reasoning MoE discriminator can supply disentangled corrective signals that steer the policy toward the supervision distribution without access to teacher logits or internal model states.","pith_extraction_headline":"Inserting a black-box on-policy distillation stage after SFT corrects distributional drift and raises final multimodal RL accuracy."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.28123/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-20T20:40:33.621213Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T18:36:34.360289Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"6dc8a610ed9582b0e08a9fa415d474f10a735e4333c5caf04a30b292224f712b"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"e297ab29-1367-4c9d-9392-48e573dc728f"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-30T02:17:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"KjXlMNwlbPjleK76RIGqMJFoHjaFOiVHMnje+RsbgOz+PR+b6OgdkpbtcZhd8/MmM6LBemlRskMh1FGKVzHCDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T14:44:49.958997Z"},"content_sha256":"117637bc9791b822caa880db00d048fcec72e3c9278c422c48dfc798e10bddbc","schema_version":"1.0","event_id":"sha256:117637bc9791b822caa880db00d048fcec72e3c9278c422c48dfc798e10bddbc"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/AP2LEEU2QSIM5A7PWCFVG2LX7H/bundle.json","state_url":"https://pith.science/pith/AP2LEEU2QSIM5A7PWCFVG2LX7H/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/AP2LEEU2QSIM5A7PWCFVG2LX7H/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-12T14:44:49Z","links":{"resolver":"https://pith.science/pith/AP2LEEU2QSIM5A7PWCFVG2LX7H","bundle":"https://pith.science/pith/AP2LEEU2QSIM5A7PWCFVG2LX7H/bundle.json","state":"https://pith.science/pith/AP2LEEU2QSIM5A7PWCFVG2LX7H/state.json","well_known_bundle":"https://pith.science/.well-known/pith/AP2LEEU2QSIM5A7PWCFVG2LX7H/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:AP2LEEU2QSIM5A7PWCFVG2LX7H","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e3c97e22801ea9e20c07c0c9a43cfe0bc7a0a8791f6969bf1a0a9909cf132dc4","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-04-30T17:12:53Z","title_canon_sha256":"27c5ef591869b652c48eab61e41decb472da3f207ba814ae0c9c6f06475994c0"},"schema_version":"1.0","source":{"id":"2604.28123","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.28123","created_at":"2026-06-30T02:17:21Z"},{"alias_kind":"arxiv_version","alias_value":"2604.28123v3","created_at":"2026-06-30T02:17:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.28123","created_at":"2026-06-30T02:17:21Z"},{"alias_kind":"pith_short_12","alias_value":"AP2LEEU2QSIM","created_at":"2026-06-30T02:17:21Z"},{"alias_kind":"pith_short_16","alias_value":"AP2LEEU2QSIM5A7P","created_at":"2026-06-30T02:17:21Z"},{"alias_kind":"pith_short_8","alias_value":"AP2LEEU2","created_at":"2026-06-30T02:17:21Z"}],"graph_snapshots":[{"event_id":"sha256:117637bc9791b822caa880db00d048fcec72e3c9278c422c48dfc798e10bddbc","target":"graph","created_at":"2026-06-30T02:17:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Experiments on Qwen3-VL show that PRISM consistently improves downstream RLVR performance across multiple RL algorithms (GRPO, DAPO, GSPO) and diverse multimodal benchmarks, improving average accuracy by +4.4 and +6.0 points over the SFT-to-RLVR baseline on 4B and 8B, respectively."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The black-box on-policy distillation game with a perception-reasoning MoE discriminator can supply disentangled corrective signals that steer the policy toward the supervision distribution without access to teacher logits or internal model states."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"PRISM adds a distribution-alignment stage using black-box on-policy distillation against a perception-reasoning MoE discriminator, yielding +4.4 and +6.0 average accuracy gains over standard SFT-to-RLVR on Qwen3-VL 4B and 8B models."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Inserting a black-box on-policy distillation stage after SFT corrects distributional drift and raises final multimodal RL accuracy."}],"snapshot_sha256":"804495cb19d534fe7931bfcca29387fdb98d2c107ee7127f8f848ced1deebe9a"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-20T20:40:33.621213Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-19T18:36:34.360289Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2604.28123/integrity.json","findings":[],"snapshot_sha256":"6dc8a610ed9582b0e08a9fa415d474f10a735e4333c5caf04a30b292224f712b","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The standard post-training recipe for large multimodal models (LMMs) applies supervised fine-tuning (SFT) on curated demonstrations followed by reinforcement learning with verifiable rewards (RLVR). However, SFT introduces distributional drift that neither preserves the model's original capabilities nor faithfully matches the supervision distribution. This problem is further amplified in multimodal reasoning, where perception errors and reasoning failures follow distinct drift patterns that compound during subsequent RL. We introduce PRISM, a three-stage pipeline that mitigates this drift by i","authors_text":"Beier Zhu, Chaojun Xiao, Chen Chen, Chengwei Qin, Hehai Lin, Keming Wu, Sudong Wang, Weiquan Huang, Wenxuan Wang, Xiaomin Yu, Yunjian Zhang, Zuhao Yang","cross_cats":["cs.AI","cs.CL"],"headline":"Inserting a black-box on-policy distillation stage after SFT corrects distributional drift and raises final multimodal RL accuracy.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-04-30T17:12:53Z","title":"Beyond SFT-to-RL: Pre-alignment via Black-Box On-Policy Distillation for Multimodal RL"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.28123","kind":"arxiv","version":3},"verdict":{"created_at":"2026-05-07T07:38:02.491032Z","id":"e297ab29-1367-4c9d-9392-48e573dc728f","model_set":{"reader":"grok-4.3"},"one_line_summary":"PRISM adds a distribution-alignment stage using black-box on-policy distillation against a perception-reasoning MoE discriminator, yielding +4.4 and +6.0 average accuracy gains over standard SFT-to-RLVR on Qwen3-VL 4B and 8B models.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Inserting a black-box on-policy distillation stage after SFT corrects distributional drift and raises final multimodal RL accuracy.","strongest_claim":"Experiments on Qwen3-VL show that PRISM consistently improves downstream RLVR performance across multiple RL algorithms (GRPO, DAPO, GSPO) and diverse multimodal benchmarks, improving average accuracy by +4.4 and +6.0 points over the SFT-to-RLVR baseline on 4B and 8B, respectively.","weakest_assumption":"The black-box on-policy distillation game with a perception-reasoning MoE discriminator can supply disentangled corrective signals that steer the policy toward the supervision distribution without access to teacher logits or internal model states."}},"verdict_id":"e297ab29-1367-4c9d-9392-48e573dc728f"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3de6b39fa11a9cfff8b82fc8e0aedc09ba2ea3dca63cd92c1d25883cda41cc23","target":"record","created_at":"2026-06-30T02:17:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e3c97e22801ea9e20c07c0c9a43cfe0bc7a0a8791f6969bf1a0a9909cf132dc4","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-04-30T17:12:53Z","title_canon_sha256":"27c5ef591869b652c48eab61e41decb472da3f207ba814ae0c9c6f06475994c0"},"schema_version":"1.0","source":{"id":"2604.28123","kind":"arxiv","version":3}},"canonical_sha256":"03f4b2129a8490ce83efb08b536977f9e2dd7015a1df6c67d3e64af7b5dd0c10","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"03f4b2129a8490ce83efb08b536977f9e2dd7015a1df6c67d3e64af7b5dd0c10","first_computed_at":"2026-06-30T02:17:21.860964Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-30T02:17:21.860964Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"LiwWNg30KQNyEN1ejnKrYYnhNITHcgK3s2M5/gClGllNMcMzmkBt9MwbWlfEgzw1feF6n5Ir36GdLgIzusZCAA==","signature_status":"signed_v1","signed_at":"2026-06-30T02:17:21.861666Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.28123","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3de6b39fa11a9cfff8b82fc8e0aedc09ba2ea3dca63cd92c1d25883cda41cc23","sha256:117637bc9791b822caa880db00d048fcec72e3c9278c422c48dfc798e10bddbc"],"state_sha256":"64633552803476ac9bc8ec416395e179b42993108f140995fd12df82498ad11a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xVMQ52fFR0+7UbtF5vB2V81rXVT1aXEq+kM4xsYuZUUbtsPqkll3ai6sW4k24DHwQstfMwWY5aqDEeQfjo85BQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-12T14:44:49.964101Z","bundle_sha256":"2b2f4e94667b46524772f1ede4053eabed52f0fd5c146ee97c5b83c3f853808f"}}