{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:F7MWPPNQ4SW3TDUEGUIQAX2CES","short_pith_number":"pith:F7MWPPNQ","schema_version":"1.0","canonical_sha256":"2fd967bdb0e4adb98e843511005f4224876c7689c14bfea72035860b847b496b","source":{"kind":"arxiv","id":"2306.01016","version":1},"attestation_state":"computed","paper":{"title":"PV2TEA: Patching Visual Modality to Textual-Established Information Extraction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.MM"],"primary_cat":"cs.CL","authors_text":"Carl Yang, Chenwei Zhang, Hejie Cui, Jingbo Shang, Nasser Zalmout, Rongmei Lin, Xian Li","submitted_at":"2023-06-01T05:39:45Z","abstract_excerpt":"Information extraction, e.g., attribute value extraction, has been extensively studied and formulated based only on text. However, many attributes can benefit from image-based extraction, like color, shape, pattern, among others. The visual modality has long been underutilized, mainly due to multimodal annotation difficulty. In this paper, we aim to patch the visual modality to the textual-established attribute information extractor. The cross-modality integration faces several unique challenges: (C1) images and textual descriptions are loosely paired intra-sample and inter-samples; (C2) image"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.01016","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-01T05:39:45Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG","cs.MM"],"title_canon_sha256":"1ef90bd95aa4c72b03001f1514c05aa2b7d9e377bf390544a714ff00b91e2ba0","abstract_canon_sha256":"04b68a45a94cb9bb026ed2a82e93f7a69c25600c06c22b89707f3175927dd794"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:16:37.063624Z","signature_b64":"lBP70cFWqdo2DCx1ufcgu7NQbdNp4ui7yb5ZBqesfXdxxuD/264Mjeh4x/JKkE62ti5IievdJ+dTXfWitI/fBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2fd967bdb0e4adb98e843511005f4224876c7689c14bfea72035860b847b496b","last_reissued_at":"2026-07-05T06:16:37.063167Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:16:37.063167Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PV2TEA: Patching Visual Modality to Textual-Established Information Extraction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.MM"],"primary_cat":"cs.CL","authors_text":"Carl Yang, Chenwei Zhang, Hejie Cui, Jingbo Shang, Nasser Zalmout, Rongmei Lin, Xian Li","submitted_at":"2023-06-01T05:39:45Z","abstract_excerpt":"Information extraction, e.g., attribute value extraction, has been extensively studied and formulated based only on text. However, many attributes can benefit from image-based extraction, like color, shape, pattern, among others. The visual modality has long been underutilized, mainly due to multimodal annotation difficulty. In this paper, we aim to patch the visual modality to the textual-established attribute information extractor. The cross-modality integration faces several unique challenges: (C1) images and textual descriptions are loosely paired intra-sample and inter-samples; (C2) image"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.01016","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.01016/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.01016","created_at":"2026-07-05T06:16:37.063241+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.01016v1","created_at":"2026-07-05T06:16:37.063241+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.01016","created_at":"2026-07-05T06:16:37.063241+00:00"},{"alias_kind":"pith_short_12","alias_value":"F7MWPPNQ4SW3","created_at":"2026-07-05T06:16:37.063241+00:00"},{"alias_kind":"pith_short_16","alias_value":"F7MWPPNQ4SW3TDUE","created_at":"2026-07-05T06:16:37.063241+00:00"},{"alias_kind":"pith_short_8","alias_value":"F7MWPPNQ","created_at":"2026-07-05T06:16:37.063241+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.00711","citing_title":"VIKSER: Visual Knowledge-Driven Self-Reinforcing Reasoning Framework","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F7MWPPNQ4SW3TDUEGUIQAX2CES","json":"https://pith.science/pith/F7MWPPNQ4SW3TDUEGUIQAX2CES.json","graph_json":"https://pith.science/api/pith-number/F7MWPPNQ4SW3TDUEGUIQAX2CES/graph.json","events_json":"https://pith.science/api/pith-number/F7MWPPNQ4SW3TDUEGUIQAX2CES/events.json","paper":"https://pith.science/paper/F7MWPPNQ"},"agent_actions":{"view_html":"https://pith.science/pith/F7MWPPNQ4SW3TDUEGUIQAX2CES","download_json":"https://pith.science/pith/F7MWPPNQ4SW3TDUEGUIQAX2CES.json","view_paper":"https://pith.science/paper/F7MWPPNQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.01016&json=true","fetch_graph":"https://pith.science/api/pith-number/F7MWPPNQ4SW3TDUEGUIQAX2CES/graph.json","fetch_events":"https://pith.science/api/pith-number/F7MWPPNQ4SW3TDUEGUIQAX2CES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F7MWPPNQ4SW3TDUEGUIQAX2CES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F7MWPPNQ4SW3TDUEGUIQAX2CES/action/storage_attestation","attest_author":"https://pith.science/pith/F7MWPPNQ4SW3TDUEGUIQAX2CES/action/author_attestation","sign_citation":"https://pith.science/pith/F7MWPPNQ4SW3TDUEGUIQAX2CES/action/citation_signature","submit_replication":"https://pith.science/pith/F7MWPPNQ4SW3TDUEGUIQAX2CES/action/replication_record"}},"created_at":"2026-07-05T06:16:37.063241+00:00","updated_at":"2026-07-05T06:16:37.063241+00:00"}