{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JAJHWQZ7D37KOY4VPFT76RGMG7","short_pith_number":"pith:JAJHWQZ7","schema_version":"1.0","canonical_sha256":"48127b433f1efea763957967ff44cc37fd3bba17c5de7e51ee112f175d42d6a0","source":{"kind":"arxiv","id":"2503.13792","version":1},"attestation_state":"computed","paper":{"title":"Identifying and Mitigating Position Bias of Multi-image Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jing Zhang, Shu Zou, Xinyu Tian, Zhaoyuan Yang","submitted_at":"2025-03-18T00:45:02Z","abstract_excerpt":"The evolution of Large Vision-Language Models (LVLMs) has progressed from single to multi-image reasoning. Despite this advancement, our findings indicate that LVLMs struggle to robustly utilize information across multiple images, with predictions significantly affected by the alteration of image positions. To further explore this issue, we introduce Position-wise Question Answering (PQA), a meticulously designed task to quantify reasoning capabilities at each position. Our analysis reveals a pronounced position bias in LVLMs: open-source models excel in reasoning with images positioned later "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.13792","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-18T00:45:02Z","cross_cats_sorted":[],"title_canon_sha256":"1ccae35c82c5f444bb7f4c9d808a139901ea728b4cd340c19cb307c611512aa8","abstract_canon_sha256":"208fda7c574c6c047dc0eab61bc88b5dc6f0a105e48b9b7aa2d4d8582806998f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:33:18.520866Z","signature_b64":"dx1pfvdcCLgXl2rPWH82Bpt+yoLrNRiu7NdYUEwzx9hhyFEDPfJdM+f7j+rvWDy7cSB+GVA1Fu95lNu3GyDdAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"48127b433f1efea763957967ff44cc37fd3bba17c5de7e51ee112f175d42d6a0","last_reissued_at":"2026-07-05T10:33:18.520105Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:33:18.520105Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Identifying and Mitigating Position Bias of Multi-image Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jing Zhang, Shu Zou, Xinyu Tian, Zhaoyuan Yang","submitted_at":"2025-03-18T00:45:02Z","abstract_excerpt":"The evolution of Large Vision-Language Models (LVLMs) has progressed from single to multi-image reasoning. Despite this advancement, our findings indicate that LVLMs struggle to robustly utilize information across multiple images, with predictions significantly affected by the alteration of image positions. To further explore this issue, we introduce Position-wise Question Answering (PQA), a meticulously designed task to quantify reasoning capabilities at each position. Our analysis reveals a pronounced position bias in LVLMs: open-source models excel in reasoning with images positioned later "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.13792","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.13792/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.13792","created_at":"2026-07-05T10:33:18.520201+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.13792v1","created_at":"2026-07-05T10:33:18.520201+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.13792","created_at":"2026-07-05T10:33:18.520201+00:00"},{"alias_kind":"pith_short_12","alias_value":"JAJHWQZ7D37K","created_at":"2026-07-05T10:33:18.520201+00:00"},{"alias_kind":"pith_short_16","alias_value":"JAJHWQZ7D37KOY4V","created_at":"2026-07-05T10:33:18.520201+00:00"},{"alias_kind":"pith_short_8","alias_value":"JAJHWQZ7","created_at":"2026-07-05T10:33:18.520201+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20731","citing_title":"TASTE: A Designer-Annotated Multi-Dimensional Preference Dataset for AI-Generated Graphic Design","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20731","citing_title":"TASTE: A Designer-Annotated Multi-Dimensional Preference Dataset for AI-Generated Graphic Design","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JAJHWQZ7D37KOY4VPFT76RGMG7","json":"https://pith.science/pith/JAJHWQZ7D37KOY4VPFT76RGMG7.json","graph_json":"https://pith.science/api/pith-number/JAJHWQZ7D37KOY4VPFT76RGMG7/graph.json","events_json":"https://pith.science/api/pith-number/JAJHWQZ7D37KOY4VPFT76RGMG7/events.json","paper":"https://pith.science/paper/JAJHWQZ7"},"agent_actions":{"view_html":"https://pith.science/pith/JAJHWQZ7D37KOY4VPFT76RGMG7","download_json":"https://pith.science/pith/JAJHWQZ7D37KOY4VPFT76RGMG7.json","view_paper":"https://pith.science/paper/JAJHWQZ7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.13792&json=true","fetch_graph":"https://pith.science/api/pith-number/JAJHWQZ7D37KOY4VPFT76RGMG7/graph.json","fetch_events":"https://pith.science/api/pith-number/JAJHWQZ7D37KOY4VPFT76RGMG7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JAJHWQZ7D37KOY4VPFT76RGMG7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JAJHWQZ7D37KOY4VPFT76RGMG7/action/storage_attestation","attest_author":"https://pith.science/pith/JAJHWQZ7D37KOY4VPFT76RGMG7/action/author_attestation","sign_citation":"https://pith.science/pith/JAJHWQZ7D37KOY4VPFT76RGMG7/action/citation_signature","submit_replication":"https://pith.science/pith/JAJHWQZ7D37KOY4VPFT76RGMG7/action/replication_record"}},"created_at":"2026-07-05T10:33:18.520201+00:00","updated_at":"2026-07-05T10:33:18.520201+00:00"}