{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZPQVTZTL2VETDWMY5FTHP3JXAH","short_pith_number":"pith:ZPQVTZTL","schema_version":"1.0","canonical_sha256":"cbe159e66bd54931d998e96677ed3701f78989e016d8f454fc6853a4d304ed3b","source":{"kind":"arxiv","id":"2410.05474","version":1},"attestation_state":"computed","paper":{"title":"R-Bench: Are your Large Multimodal Model Robust to Real-world Corruptions?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.MM","eess.IV"],"primary_cat":"cs.CV","authors_text":"Chunyi Li, Guangtao Zhai, Guo Lu, Haoning Wu, Jianbo Zhang, Weisi Lin, Wei Sun, Xiaohong Liu, Xiongkuo Min, Yuan Tian, Zicheng Zhang","submitted_at":"2024-10-07T20:12:08Z","abstract_excerpt":"The outstanding performance of Large Multimodal Models (LMMs) has made them widely applied in vision-related tasks. However, various corruptions in the real world mean that images will not be as ideal as in simulations, presenting significant challenges for the practical application of LMMs. To address this issue, we introduce R-Bench, a benchmark focused on the **Real-world Robustness of LMMs**. Specifically, we: (a) model the complete link from user capture to LMMs reception, comprising 33 corruption dimensions, including 7 steps according to the corruption sequence, and 7 groups based on lo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05474","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-07T20:12:08Z","cross_cats_sorted":["cs.MM","eess.IV"],"title_canon_sha256":"f2391b6ea6ca45205ee7c5d2e13be54e9252c8b7dddbd99b5e0a32203076da24","abstract_canon_sha256":"9d058b47a84693480b446757ea4cbaa2ccf4215b55aa8bbd7681a528af3820d3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:19.845082Z","signature_b64":"Q4IOQ4985h3eIzYQtZm8lrU9+vcQBnRE0bbGWocOkPSug1HBlWP1cXJWoGMhz8sz5boj3HrXlTC/x8H842qnCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cbe159e66bd54931d998e96677ed3701f78989e016d8f454fc6853a4d304ed3b","last_reissued_at":"2026-07-05T09:17:19.844543Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:19.844543Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"R-Bench: Are your Large Multimodal Model Robust to Real-world Corruptions?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.MM","eess.IV"],"primary_cat":"cs.CV","authors_text":"Chunyi Li, Guangtao Zhai, Guo Lu, Haoning Wu, Jianbo Zhang, Weisi Lin, Wei Sun, Xiaohong Liu, Xiongkuo Min, Yuan Tian, Zicheng Zhang","submitted_at":"2024-10-07T20:12:08Z","abstract_excerpt":"The outstanding performance of Large Multimodal Models (LMMs) has made them widely applied in vision-related tasks. However, various corruptions in the real world mean that images will not be as ideal as in simulations, presenting significant challenges for the practical application of LMMs. To address this issue, we introduce R-Bench, a benchmark focused on the **Real-world Robustness of LMMs**. Specifically, we: (a) model the complete link from user capture to LMMs reception, comprising 33 corruption dimensions, including 7 steps according to the corruption sequence, and 7 groups based on lo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05474","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05474/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05474","created_at":"2026-07-05T09:17:19.844641+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05474v1","created_at":"2026-07-05T09:17:19.844641+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05474","created_at":"2026-07-05T09:17:19.844641+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZPQVTZTL2VET","created_at":"2026-07-05T09:17:19.844641+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZPQVTZTL2VETDWMY","created_at":"2026-07-05T09:17:19.844641+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZPQVTZTL","created_at":"2026-07-05T09:17:19.844641+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07687","citing_title":"What Makes Video World Model Latents Action-Relevant: Prediction over Reconstruction","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04780","citing_title":"CLEAR: Unlocking Generative Potential for Degraded Image Understanding in Unified Multimodal Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10479","citing_title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2508.18265","citing_title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZPQVTZTL2VETDWMY5FTHP3JXAH","json":"https://pith.science/pith/ZPQVTZTL2VETDWMY5FTHP3JXAH.json","graph_json":"https://pith.science/api/pith-number/ZPQVTZTL2VETDWMY5FTHP3JXAH/graph.json","events_json":"https://pith.science/api/pith-number/ZPQVTZTL2VETDWMY5FTHP3JXAH/events.json","paper":"https://pith.science/paper/ZPQVTZTL"},"agent_actions":{"view_html":"https://pith.science/pith/ZPQVTZTL2VETDWMY5FTHP3JXAH","download_json":"https://pith.science/pith/ZPQVTZTL2VETDWMY5FTHP3JXAH.json","view_paper":"https://pith.science/paper/ZPQVTZTL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05474&json=true","fetch_graph":"https://pith.science/api/pith-number/ZPQVTZTL2VETDWMY5FTHP3JXAH/graph.json","fetch_events":"https://pith.science/api/pith-number/ZPQVTZTL2VETDWMY5FTHP3JXAH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZPQVTZTL2VETDWMY5FTHP3JXAH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZPQVTZTL2VETDWMY5FTHP3JXAH/action/storage_attestation","attest_author":"https://pith.science/pith/ZPQVTZTL2VETDWMY5FTHP3JXAH/action/author_attestation","sign_citation":"https://pith.science/pith/ZPQVTZTL2VETDWMY5FTHP3JXAH/action/citation_signature","submit_replication":"https://pith.science/pith/ZPQVTZTL2VETDWMY5FTHP3JXAH/action/replication_record"}},"created_at":"2026-07-05T09:17:19.844641+00:00","updated_at":"2026-07-05T09:17:19.844641+00:00"}