{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HOIZPWWTXCAVSUIM57P4M4VK4K","short_pith_number":"pith:HOIZPWWT","schema_version":"1.0","canonical_sha256":"3b9197dad3b88159510cefdfc672aae282f195f71ca0ac16305d0d177b45a639","source":{"kind":"arxiv","id":"2406.00980","version":1},"attestation_state":"computed","paper":{"title":"Selectively Answering Visual Questions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Guido Ivetta, Hern\\'an Maina, Julian Martin Eisenschlos, Luciana Benotti","submitted_at":"2024-06-03T04:28:10Z","abstract_excerpt":"Recently, large multi-modal models (LMMs) have emerged with the capacity to perform vision tasks such as captioning and visual question answering (VQA) with unprecedented accuracy. Applications such as helping the blind or visually impaired have a critical need for precise answers. It is specially important for models to be well calibrated and be able to quantify their uncertainty in order to selectively decide when to answer and when to abstain or ask for clarifications. We perform the first in-depth analysis of calibration methods and metrics for VQA with in-context learning LMMs. Studying V"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.00980","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-03T04:28:10Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"d7518157c8173b6423e6c87349e001904e84ea2dd15cbcff070f385c731a1526","abstract_canon_sha256":"3e10a15934f7d9baaee7e93c0b6da5d811f888b58f6c7d4bdbb05d5a004f1ced"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:26:33.426925Z","signature_b64":"kVQc1uQ5EVVIgGTIEAMCe2eCXsZnazighbNr2cMa84JI5D8mAH85T2gXiyxsWuC7cby6PcT/S/3sK4mkv8FJAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3b9197dad3b88159510cefdfc672aae282f195f71ca0ac16305d0d177b45a639","last_reissued_at":"2026-07-05T08:26:33.426539Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:26:33.426539Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Selectively Answering Visual Questions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Guido Ivetta, Hern\\'an Maina, Julian Martin Eisenschlos, Luciana Benotti","submitted_at":"2024-06-03T04:28:10Z","abstract_excerpt":"Recently, large multi-modal models (LMMs) have emerged with the capacity to perform vision tasks such as captioning and visual question answering (VQA) with unprecedented accuracy. Applications such as helping the blind or visually impaired have a critical need for precise answers. It is specially important for models to be well calibrated and be able to quantify their uncertainty in order to selectively decide when to answer and when to abstain or ask for clarifications. We perform the first in-depth analysis of calibration methods and metrics for VQA with in-context learning LMMs. Studying V"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.00980","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.00980/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.00980","created_at":"2026-07-05T08:26:33.426598+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.00980v1","created_at":"2026-07-05T08:26:33.426598+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.00980","created_at":"2026-07-05T08:26:33.426598+00:00"},{"alias_kind":"pith_short_12","alias_value":"HOIZPWWTXCAV","created_at":"2026-07-05T08:26:33.426598+00:00"},{"alias_kind":"pith_short_16","alias_value":"HOIZPWWTXCAVSUIM","created_at":"2026-07-05T08:26:33.426598+00:00"},{"alias_kind":"pith_short_8","alias_value":"HOIZPWWT","created_at":"2026-07-05T08:26:33.426598+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HOIZPWWTXCAVSUIM57P4M4VK4K","json":"https://pith.science/pith/HOIZPWWTXCAVSUIM57P4M4VK4K.json","graph_json":"https://pith.science/api/pith-number/HOIZPWWTXCAVSUIM57P4M4VK4K/graph.json","events_json":"https://pith.science/api/pith-number/HOIZPWWTXCAVSUIM57P4M4VK4K/events.json","paper":"https://pith.science/paper/HOIZPWWT"},"agent_actions":{"view_html":"https://pith.science/pith/HOIZPWWTXCAVSUIM57P4M4VK4K","download_json":"https://pith.science/pith/HOIZPWWTXCAVSUIM57P4M4VK4K.json","view_paper":"https://pith.science/paper/HOIZPWWT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.00980&json=true","fetch_graph":"https://pith.science/api/pith-number/HOIZPWWTXCAVSUIM57P4M4VK4K/graph.json","fetch_events":"https://pith.science/api/pith-number/HOIZPWWTXCAVSUIM57P4M4VK4K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HOIZPWWTXCAVSUIM57P4M4VK4K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HOIZPWWTXCAVSUIM57P4M4VK4K/action/storage_attestation","attest_author":"https://pith.science/pith/HOIZPWWTXCAVSUIM57P4M4VK4K/action/author_attestation","sign_citation":"https://pith.science/pith/HOIZPWWTXCAVSUIM57P4M4VK4K/action/citation_signature","submit_replication":"https://pith.science/pith/HOIZPWWTXCAVSUIM57P4M4VK4K/action/replication_record"}},"created_at":"2026-07-05T08:26:33.426598+00:00","updated_at":"2026-07-05T08:26:33.426598+00:00"}