{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FRBMYUTO7ZD7VWAIBVS56RI6MQ","short_pith_number":"pith:FRBMYUTO","schema_version":"1.0","canonical_sha256":"2c42cc526efe47fad8080d65df451e641395347e629bc99675173a8e0ba932a5","source":{"kind":"arxiv","id":"2406.08100","version":1},"attestation_state":"computed","paper":{"title":"Multimodal Table Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Mingyu Zheng, Qiaoqiao She, Qingyi Si, Weiping Wang, Wenbin Jiang, Xinwei Feng, Zheng Lin","submitted_at":"2024-06-12T11:27:03Z","abstract_excerpt":"Although great progress has been made by previous table understanding methods including recent approaches based on large language models (LLMs), they rely heavily on the premise that given tables must be converted into a certain text sequence (such as Markdown or HTML) to serve as model input. However, it is difficult to access such high-quality textual table representations in some real-world scenarios, and table images are much more accessible. Therefore, how to directly understand tables using intuitive visual information is a crucial and urgent challenge for developing more practical appli"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08100","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-12T11:27:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c31d540aae2e706c84d1ed6a1019ff6aa41ce25e4277709d22e0a81935c80be8","abstract_canon_sha256":"1d9200fb1aba7afa51382afc90cc165e94570e00d725388918e6df81f216e7cf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:54.490423Z","signature_b64":"VyIv7AFx79+Ik2upq3q0bqPzdAck/ejXyaadfLv4ZHWTLAj6oco/JkHb0nuwMH5SqhDnINaZGPIcz+MOsPs1CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c42cc526efe47fad8080d65df451e641395347e629bc99675173a8e0ba932a5","last_reissued_at":"2026-07-05T08:30:54.489931Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:54.489931Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal Table Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Mingyu Zheng, Qiaoqiao She, Qingyi Si, Weiping Wang, Wenbin Jiang, Xinwei Feng, Zheng Lin","submitted_at":"2024-06-12T11:27:03Z","abstract_excerpt":"Although great progress has been made by previous table understanding methods including recent approaches based on large language models (LLMs), they rely heavily on the premise that given tables must be converted into a certain text sequence (such as Markdown or HTML) to serve as model input. However, it is difficult to access such high-quality textual table representations in some real-world scenarios, and table images are much more accessible. Therefore, how to directly understand tables using intuitive visual information is a crucial and urgent challenge for developing more practical appli"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08100","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08100/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08100","created_at":"2026-07-05T08:30:54.489990+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08100v1","created_at":"2026-07-05T08:30:54.489990+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08100","created_at":"2026-07-05T08:30:54.489990+00:00"},{"alias_kind":"pith_short_12","alias_value":"FRBMYUTO7ZD7","created_at":"2026-07-05T08:30:54.489990+00:00"},{"alias_kind":"pith_short_16","alias_value":"FRBMYUTO7ZD7VWAI","created_at":"2026-07-05T08:30:54.489990+00:00"},{"alias_kind":"pith_short_8","alias_value":"FRBMYUTO","created_at":"2026-07-05T08:30:54.489990+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.22693","citing_title":"Bridging Language Models and Financial Analysis","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08146","citing_title":"VT-Bench: A Unified Benchmark for Visual-Tabular Multi-Modal Learning","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16099","citing_title":"DenTab: A Dataset for Table Recognition and Visual QA on Real-World Dental Estimates","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FRBMYUTO7ZD7VWAIBVS56RI6MQ","json":"https://pith.science/pith/FRBMYUTO7ZD7VWAIBVS56RI6MQ.json","graph_json":"https://pith.science/api/pith-number/FRBMYUTO7ZD7VWAIBVS56RI6MQ/graph.json","events_json":"https://pith.science/api/pith-number/FRBMYUTO7ZD7VWAIBVS56RI6MQ/events.json","paper":"https://pith.science/paper/FRBMYUTO"},"agent_actions":{"view_html":"https://pith.science/pith/FRBMYUTO7ZD7VWAIBVS56RI6MQ","download_json":"https://pith.science/pith/FRBMYUTO7ZD7VWAIBVS56RI6MQ.json","view_paper":"https://pith.science/paper/FRBMYUTO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08100&json=true","fetch_graph":"https://pith.science/api/pith-number/FRBMYUTO7ZD7VWAIBVS56RI6MQ/graph.json","fetch_events":"https://pith.science/api/pith-number/FRBMYUTO7ZD7VWAIBVS56RI6MQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FRBMYUTO7ZD7VWAIBVS56RI6MQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FRBMYUTO7ZD7VWAIBVS56RI6MQ/action/storage_attestation","attest_author":"https://pith.science/pith/FRBMYUTO7ZD7VWAIBVS56RI6MQ/action/author_attestation","sign_citation":"https://pith.science/pith/FRBMYUTO7ZD7VWAIBVS56RI6MQ/action/citation_signature","submit_replication":"https://pith.science/pith/FRBMYUTO7ZD7VWAIBVS56RI6MQ/action/replication_record"}},"created_at":"2026-07-05T08:30:54.489990+00:00","updated_at":"2026-07-05T08:30:54.489990+00:00"}