{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HLPL2NKTC64AWOKCZZL4NE2DP3","short_pith_number":"pith:HLPL2NKT","schema_version":"1.0","canonical_sha256":"3adebd355317b80b3942ce57c693437eeb9d6d0c5d362f5bf5373c3ad62907d3","source":{"kind":"arxiv","id":"2405.08807","version":2},"attestation_state":"computed","paper":{"title":"SciFIBench: Benchmarking Large Multimodal Models for Scientific Figure Interpretation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jonathan Roberts, Kai Han, Neil Houlsby, Samuel Albanie","submitted_at":"2024-05-14T17:54:17Z","abstract_excerpt":"Large multimodal models (LMMs) have proven flexible and generalisable across many tasks and fields. Although they have strong potential to aid scientific research, their capabilities in this domain are not well characterised. A key aspect of scientific research is the ability to understand and interpret figures, which serve as a rich, compressed source of complex information. In this work, we present SciFIBench, a scientific figure interpretation benchmark consisting of 2000 questions split between two tasks across 8 categories. The questions are curated from arXiv paper figures and captions, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.08807","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-14T17:54:17Z","cross_cats_sorted":[],"title_canon_sha256":"9408100f18268218806fce90559953e5af1880e6d8ba0b2cb098ba0eeaad8a37","abstract_canon_sha256":"59f5f8c725e677061dad282d2cd014c10e3e033187aeeef5a546a264fdca7c84"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:44:50.662079Z","signature_b64":"/Dg25P6LYaQvXJcVYRxiImQolv6PF5+ZPhqj15zjlXDCX/dJs4IJljWzTmMkUEOyhi+J8iXOuaaAFzdXl5g/AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3adebd355317b80b3942ce57c693437eeb9d6d0c5d362f5bf5373c3ad62907d3","last_reissued_at":"2026-07-05T09:44:50.661524Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:44:50.661524Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SciFIBench: Benchmarking Large Multimodal Models for Scientific Figure Interpretation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jonathan Roberts, Kai Han, Neil Houlsby, Samuel Albanie","submitted_at":"2024-05-14T17:54:17Z","abstract_excerpt":"Large multimodal models (LMMs) have proven flexible and generalisable across many tasks and fields. Although they have strong potential to aid scientific research, their capabilities in this domain are not well characterised. A key aspect of scientific research is the ability to understand and interpret figures, which serve as a rich, compressed source of complex information. In this work, we present SciFIBench, a scientific figure interpretation benchmark consisting of 2000 questions split between two tasks across 8 categories. The questions are curated from arXiv paper figures and captions, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.08807","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.08807/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.08807","created_at":"2026-07-05T09:44:50.661593+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.08807v2","created_at":"2026-07-05T09:44:50.661593+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.08807","created_at":"2026-07-05T09:44:50.661593+00:00"},{"alias_kind":"pith_short_12","alias_value":"HLPL2NKTC64A","created_at":"2026-07-05T09:44:50.661593+00:00"},{"alias_kind":"pith_short_16","alias_value":"HLPL2NKTC64AWOKC","created_at":"2026-07-05T09:44:50.661593+00:00"},{"alias_kind":"pith_short_8","alias_value":"HLPL2NKT","created_at":"2026-07-05T09:44:50.661593+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05744","citing_title":"PlanBench-V: A Spatial Planning Map Benchmark for Vision-Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04172","citing_title":"GENFIG1: Visual Summaries of Scholarly Work as a Challenge for Vision-Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2502.18864","citing_title":"Towards an AI co-scientist","ref_index":286,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08211","citing_title":"SciFigDetect: A Benchmark for AI-Generated Scientific Figure Detection","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HLPL2NKTC64AWOKCZZL4NE2DP3","json":"https://pith.science/pith/HLPL2NKTC64AWOKCZZL4NE2DP3.json","graph_json":"https://pith.science/api/pith-number/HLPL2NKTC64AWOKCZZL4NE2DP3/graph.json","events_json":"https://pith.science/api/pith-number/HLPL2NKTC64AWOKCZZL4NE2DP3/events.json","paper":"https://pith.science/paper/HLPL2NKT"},"agent_actions":{"view_html":"https://pith.science/pith/HLPL2NKTC64AWOKCZZL4NE2DP3","download_json":"https://pith.science/pith/HLPL2NKTC64AWOKCZZL4NE2DP3.json","view_paper":"https://pith.science/paper/HLPL2NKT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.08807&json=true","fetch_graph":"https://pith.science/api/pith-number/HLPL2NKTC64AWOKCZZL4NE2DP3/graph.json","fetch_events":"https://pith.science/api/pith-number/HLPL2NKTC64AWOKCZZL4NE2DP3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HLPL2NKTC64AWOKCZZL4NE2DP3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HLPL2NKTC64AWOKCZZL4NE2DP3/action/storage_attestation","attest_author":"https://pith.science/pith/HLPL2NKTC64AWOKCZZL4NE2DP3/action/author_attestation","sign_citation":"https://pith.science/pith/HLPL2NKTC64AWOKCZZL4NE2DP3/action/citation_signature","submit_replication":"https://pith.science/pith/HLPL2NKTC64AWOKCZZL4NE2DP3/action/replication_record"}},"created_at":"2026-07-05T09:44:50.661593+00:00","updated_at":"2026-07-05T09:44:50.661593+00:00"}