{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PHBGAU2H5OEBOE4HQW6SIDOBPD","short_pith_number":"pith:PHBGAU2H","schema_version":"1.0","canonical_sha256":"79c2605347eb8817138785bd240dc178eccfa4fe5599f9f0b5b66518f836e082","source":{"kind":"arxiv","id":"2505.23224","version":3},"attestation_state":"computed","paper":{"title":"MMBoundary: Advancing MLLM Knowledge Boundary Awareness through Reasoning Step Confidence Calibration","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Sandeep Polisetty, Shujin Wu, Yi R. Fung, Yuchen Huang, Zhitao He, Zhiyuan Fan","submitted_at":"2025-05-29T08:14:40Z","abstract_excerpt":"In recent years, multimodal large language models (MLLMs) have made significant progress but continue to face inherent challenges in multimodal reasoning, which requires multi-level (e.g., perception, reasoning) and multi-granular (e.g., multi-step reasoning chain) advanced inferencing. Prior work on estimating model confidence tends to focus on the overall response for training and calibration, but fails to assess confidence in each reasoning step, leading to undesirable hallucination snowballing. In this work, we present MMBoundary, a novel framework that advances the knowledge boundary awar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23224","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T08:14:40Z","cross_cats_sorted":[],"title_canon_sha256":"7684d982ae73cf2c9eb85476b13c782ba4ca53515c16ce8154b8fa562805c181","abstract_canon_sha256":"737120d0df975736547f21d40fcd46fd21bde6347bb9efa3fb88c3750389c07b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:16.988523Z","signature_b64":"PUwjwzC5jkNvWJiQg6JPSRyATKfiPlPZQRlzWA/Xfy1ySlQm6kSbSa0rZ8c72jRbisXgaTYREXxGDL+d2ZbXAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"79c2605347eb8817138785bd240dc178eccfa4fe5599f9f0b5b66518f836e082","last_reissued_at":"2026-07-05T11:28:16.988022Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:16.988022Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMBoundary: Advancing MLLM Knowledge Boundary Awareness through Reasoning Step Confidence Calibration","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Sandeep Polisetty, Shujin Wu, Yi R. Fung, Yuchen Huang, Zhitao He, Zhiyuan Fan","submitted_at":"2025-05-29T08:14:40Z","abstract_excerpt":"In recent years, multimodal large language models (MLLMs) have made significant progress but continue to face inherent challenges in multimodal reasoning, which requires multi-level (e.g., perception, reasoning) and multi-granular (e.g., multi-step reasoning chain) advanced inferencing. Prior work on estimating model confidence tends to focus on the overall response for training and calibration, but fails to assess confidence in each reasoning step, leading to undesirable hallucination snowballing. In this work, we present MMBoundary, a novel framework that advances the knowledge boundary awar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23224","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23224/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23224","created_at":"2026-07-05T11:28:16.988078+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23224v3","created_at":"2026-07-05T11:28:16.988078+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23224","created_at":"2026-07-05T11:28:16.988078+00:00"},{"alias_kind":"pith_short_12","alias_value":"PHBGAU2H5OEB","created_at":"2026-07-05T11:28:16.988078+00:00"},{"alias_kind":"pith_short_16","alias_value":"PHBGAU2H5OEBOE4H","created_at":"2026-07-05T11:28:16.988078+00:00"},{"alias_kind":"pith_short_8","alias_value":"PHBGAU2H","created_at":"2026-07-05T11:28:16.988078+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":257,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PHBGAU2H5OEBOE4HQW6SIDOBPD","json":"https://pith.science/pith/PHBGAU2H5OEBOE4HQW6SIDOBPD.json","graph_json":"https://pith.science/api/pith-number/PHBGAU2H5OEBOE4HQW6SIDOBPD/graph.json","events_json":"https://pith.science/api/pith-number/PHBGAU2H5OEBOE4HQW6SIDOBPD/events.json","paper":"https://pith.science/paper/PHBGAU2H"},"agent_actions":{"view_html":"https://pith.science/pith/PHBGAU2H5OEBOE4HQW6SIDOBPD","download_json":"https://pith.science/pith/PHBGAU2H5OEBOE4HQW6SIDOBPD.json","view_paper":"https://pith.science/paper/PHBGAU2H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23224&json=true","fetch_graph":"https://pith.science/api/pith-number/PHBGAU2H5OEBOE4HQW6SIDOBPD/graph.json","fetch_events":"https://pith.science/api/pith-number/PHBGAU2H5OEBOE4HQW6SIDOBPD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PHBGAU2H5OEBOE4HQW6SIDOBPD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PHBGAU2H5OEBOE4HQW6SIDOBPD/action/storage_attestation","attest_author":"https://pith.science/pith/PHBGAU2H5OEBOE4HQW6SIDOBPD/action/author_attestation","sign_citation":"https://pith.science/pith/PHBGAU2H5OEBOE4HQW6SIDOBPD/action/citation_signature","submit_replication":"https://pith.science/pith/PHBGAU2H5OEBOE4HQW6SIDOBPD/action/replication_record"}},"created_at":"2026-07-05T11:28:16.988078+00:00","updated_at":"2026-07-05T11:28:16.988078+00:00"}