{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XMZUPGYHJ7WSK4JWM4EPBI6M2H","short_pith_number":"pith:XMZUPGYH","schema_version":"1.0","canonical_sha256":"bb33479b074fed2571366708f0a3ccd1eb09ca6d4d7122fdbf4961dde6ea3fa6","source":{"kind":"arxiv","id":"2401.07529","version":3},"attestation_state":"computed","paper":{"title":"MM-SAP: A Comprehensive Benchmark for Assessing Self-Awareness of Multimodal Large Language Models in Perception","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Heyang Liu, Hongcheng Liu, Yanfeng Wang, Yuhao Wang, Yusheng Liao, Yu Wang","submitted_at":"2024-01-15T08:19:22Z","abstract_excerpt":"Recent advancements in Multimodal Large Language Models (MLLMs) have demonstrated exceptional capabilities in visual perception and understanding. However, these models also suffer from hallucinations, which limit their reliability as AI systems. We believe that these hallucinations are partially due to the models' struggle with understanding what they can and cannot perceive from images, a capability we refer to as self-awareness in perception. Despite its importance, this aspect of MLLMs has been overlooked in prior studies. In this paper, we aim to define and evaluate the self-awareness of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.07529","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-15T08:19:22Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"e5578715cd8f63a305ce47d5ba8732f4cdcca6ed479886f6bf4bf755e4b54ea9","abstract_canon_sha256":"6994687dfff94a1154094ed2eaf4537deaf2e6fcfee9b5351c0740421d4c9244"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:51.732993Z","signature_b64":"uBvsDTZ+U8O0clmJWH1O0COmjGWNX4SPWqEJWGQUI8lDGW2Ezv//K0LBjqrXTWJqDAQhZHfFZ4speHNXAflxDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb33479b074fed2571366708f0a3ccd1eb09ca6d4d7122fdbf4961dde6ea3fa6","last_reissued_at":"2026-07-05T08:25:51.732486Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:51.732486Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MM-SAP: A Comprehensive Benchmark for Assessing Self-Awareness of Multimodal Large Language Models in Perception","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Heyang Liu, Hongcheng Liu, Yanfeng Wang, Yuhao Wang, Yusheng Liao, Yu Wang","submitted_at":"2024-01-15T08:19:22Z","abstract_excerpt":"Recent advancements in Multimodal Large Language Models (MLLMs) have demonstrated exceptional capabilities in visual perception and understanding. However, these models also suffer from hallucinations, which limit their reliability as AI systems. We believe that these hallucinations are partially due to the models' struggle with understanding what they can and cannot perceive from images, a capability we refer to as self-awareness in perception. Despite its importance, this aspect of MLLMs has been overlooked in prior studies. In this paper, we aim to define and evaluate the self-awareness of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.07529","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.07529/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.07529","created_at":"2026-07-05T08:25:51.732548+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.07529v3","created_at":"2026-07-05T08:25:51.732548+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.07529","created_at":"2026-07-05T08:25:51.732548+00:00"},{"alias_kind":"pith_short_12","alias_value":"XMZUPGYHJ7WS","created_at":"2026-07-05T08:25:51.732548+00:00"},{"alias_kind":"pith_short_16","alias_value":"XMZUPGYHJ7WSK4JW","created_at":"2026-07-05T08:25:51.732548+00:00"},{"alias_kind":"pith_short_8","alias_value":"XMZUPGYH","created_at":"2026-07-05T08:25:51.732548+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2404.13076","citing_title":"LLM Evaluators Recognize and Favor Their Own Generations","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22884","citing_title":"Can Multimodal Large Language Models Truly Understand Small Objects?","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XMZUPGYHJ7WSK4JWM4EPBI6M2H","json":"https://pith.science/pith/XMZUPGYHJ7WSK4JWM4EPBI6M2H.json","graph_json":"https://pith.science/api/pith-number/XMZUPGYHJ7WSK4JWM4EPBI6M2H/graph.json","events_json":"https://pith.science/api/pith-number/XMZUPGYHJ7WSK4JWM4EPBI6M2H/events.json","paper":"https://pith.science/paper/XMZUPGYH"},"agent_actions":{"view_html":"https://pith.science/pith/XMZUPGYHJ7WSK4JWM4EPBI6M2H","download_json":"https://pith.science/pith/XMZUPGYHJ7WSK4JWM4EPBI6M2H.json","view_paper":"https://pith.science/paper/XMZUPGYH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.07529&json=true","fetch_graph":"https://pith.science/api/pith-number/XMZUPGYHJ7WSK4JWM4EPBI6M2H/graph.json","fetch_events":"https://pith.science/api/pith-number/XMZUPGYHJ7WSK4JWM4EPBI6M2H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XMZUPGYHJ7WSK4JWM4EPBI6M2H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XMZUPGYHJ7WSK4JWM4EPBI6M2H/action/storage_attestation","attest_author":"https://pith.science/pith/XMZUPGYHJ7WSK4JWM4EPBI6M2H/action/author_attestation","sign_citation":"https://pith.science/pith/XMZUPGYHJ7WSK4JWM4EPBI6M2H/action/citation_signature","submit_replication":"https://pith.science/pith/XMZUPGYHJ7WSK4JWM4EPBI6M2H/action/replication_record"}},"created_at":"2026-07-05T08:25:51.732548+00:00","updated_at":"2026-07-05T08:25:51.732548+00:00"}