{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WPTL5G3OCS64ALTRZRRIRZQDHI","short_pith_number":"pith:WPTL5G3O","schema_version":"1.0","canonical_sha256":"b3e6be9b6e14bdc02e71cc6288e6033a155dfc6ca26dd7a095f54f4840bdd418","source":{"kind":"arxiv","id":"2403.08730","version":2},"attestation_state":"computed","paper":{"title":"Strengthening Multimodal Large Language Model with Bootstrapped Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Jipeng Zhang, Renjie Pi, Rui Pan, Runtao Liu, Tianyang Han, Tong Zhang, Wei Xiong","submitted_at":"2024-03-13T17:29:45Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) excel in generating responses based on visual inputs. However, they often suffer from a bias towards generating responses similar to their pretraining corpus, overshadowing the importance of visual information. We treat this bias as a \"preference\" for pretraining statistics, which hinders the model's grounding in visual input. To mitigate this issue, we propose Bootstrapped Preference Optimization (BPO), which conducts preference learning with datasets containing negative responses bootstrapped from the model itself. Specifically, we propose the followi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.08730","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-13T17:29:45Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"4ce146cc93c8cd68872c816300b4623fa9f8462e8331782bd635fe998d45ba88","abstract_canon_sha256":"5f3075433cbfa75888ec2293bc60208c0b81d3538efa1f1ab580b5c583b96310"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:03:49.900724Z","signature_b64":"7niVCiYiYy++hlV3rsVTAU362jYFNzgwk3yhz9zvz4/2Av/YczIs48aoNPMz3uw0Pbv32krF47p+diZCRI0YAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b3e6be9b6e14bdc02e71cc6288e6033a155dfc6ca26dd7a095f54f4840bdd418","last_reissued_at":"2026-07-05T08:03:49.900114Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:03:49.900114Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Strengthening Multimodal Large Language Model with Bootstrapped Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Jipeng Zhang, Renjie Pi, Rui Pan, Runtao Liu, Tianyang Han, Tong Zhang, Wei Xiong","submitted_at":"2024-03-13T17:29:45Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) excel in generating responses based on visual inputs. However, they often suffer from a bias towards generating responses similar to their pretraining corpus, overshadowing the importance of visual information. We treat this bias as a \"preference\" for pretraining statistics, which hinders the model's grounding in visual input. To mitigate this issue, we propose Bootstrapped Preference Optimization (BPO), which conducts preference learning with datasets containing negative responses bootstrapped from the model itself. Specifically, we propose the followi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.08730","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.08730/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.08730","created_at":"2026-07-05T08:03:49.900185+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.08730v2","created_at":"2026-07-05T08:03:49.900185+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.08730","created_at":"2026-07-05T08:03:49.900185+00:00"},{"alias_kind":"pith_short_12","alias_value":"WPTL5G3OCS64","created_at":"2026-07-05T08:03:49.900185+00:00"},{"alias_kind":"pith_short_16","alias_value":"WPTL5G3OCS64ALTR","created_at":"2026-07-05T08:03:49.900185+00:00"},{"alias_kind":"pith_short_8","alias_value":"WPTL5G3O","created_at":"2026-07-05T08:03:49.900185+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15300","citing_title":"Deep Pre-Alignment for VLMs","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2510.21122","citing_title":"NoisyGRPO: Incentivizing Multimodal CoT Reasoning via Noise Injection and Bayesian Estimation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2407.03320","citing_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10442","citing_title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","ref_index":77,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WPTL5G3OCS64ALTRZRRIRZQDHI","json":"https://pith.science/pith/WPTL5G3OCS64ALTRZRRIRZQDHI.json","graph_json":"https://pith.science/api/pith-number/WPTL5G3OCS64ALTRZRRIRZQDHI/graph.json","events_json":"https://pith.science/api/pith-number/WPTL5G3OCS64ALTRZRRIRZQDHI/events.json","paper":"https://pith.science/paper/WPTL5G3O"},"agent_actions":{"view_html":"https://pith.science/pith/WPTL5G3OCS64ALTRZRRIRZQDHI","download_json":"https://pith.science/pith/WPTL5G3OCS64ALTRZRRIRZQDHI.json","view_paper":"https://pith.science/paper/WPTL5G3O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.08730&json=true","fetch_graph":"https://pith.science/api/pith-number/WPTL5G3OCS64ALTRZRRIRZQDHI/graph.json","fetch_events":"https://pith.science/api/pith-number/WPTL5G3OCS64ALTRZRRIRZQDHI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WPTL5G3OCS64ALTRZRRIRZQDHI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WPTL5G3OCS64ALTRZRRIRZQDHI/action/storage_attestation","attest_author":"https://pith.science/pith/WPTL5G3OCS64ALTRZRRIRZQDHI/action/author_attestation","sign_citation":"https://pith.science/pith/WPTL5G3OCS64ALTRZRRIRZQDHI/action/citation_signature","submit_replication":"https://pith.science/pith/WPTL5G3OCS64ALTRZRRIRZQDHI/action/replication_record"}},"created_at":"2026-07-05T08:03:49.900185+00:00","updated_at":"2026-07-05T08:03:49.900185+00:00"}