{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VRJL3XBMC5ZPYKHFXGEC6ZKMFV","short_pith_number":"pith:VRJL3XBM","schema_version":"1.0","canonical_sha256":"ac52bddc2c1772fc28e5b9882f654c2d708662075dcac60478ad2e5345217b88","source":{"kind":"arxiv","id":"2309.04734","version":1},"attestation_state":"computed","paper":{"title":"Towards Better Multi-modal Keyphrase Generation via Visual Entity Enhancement and Multi-granularity Image Noise Filtering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Fandong Meng, Jianxin Lin, Jie Zhou, Jinsong Su, Suhang Wu, Xiaoli Wang, Yifan Dong","submitted_at":"2023-09-09T09:41:36Z","abstract_excerpt":"Multi-modal keyphrase generation aims to produce a set of keyphrases that represent the core points of the input text-image pair. In this regard, dominant methods mainly focus on multi-modal fusion for keyphrase generation. Nevertheless, there are still two main drawbacks: 1) only a limited number of sources, such as image captions, can be utilized to provide auxiliary information. However, they may not be sufficient for the subsequent keyphrase generation. 2) the input text and image are often not perfectly matched, and thus the image may introduce noise into the model. To address these limit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.04734","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-09-09T09:41:36Z","cross_cats_sorted":["cs.CL","cs.MM"],"title_canon_sha256":"93668d42e0823440a3a8f7f8343fb4bd88e15fb903e84fa6478823dff0b71f2d","abstract_canon_sha256":"742e526bc3402bc22209239c68ce4b6251ad347e6286adfbe8cc3fefa8d80373"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:49:08.029615Z","signature_b64":"g73xhs460+sDRdyU4ddPQ7CKc88YF7csNNO+efCpBONeBy+Lz9UqGrgbyCB8e9nqC6wWyNbl3xq4d8H+S67pCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac52bddc2c1772fc28e5b9882f654c2d708662075dcac60478ad2e5345217b88","last_reissued_at":"2026-07-05T06:49:08.029134Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:49:08.029134Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Better Multi-modal Keyphrase Generation via Visual Entity Enhancement and Multi-granularity Image Noise Filtering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Fandong Meng, Jianxin Lin, Jie Zhou, Jinsong Su, Suhang Wu, Xiaoli Wang, Yifan Dong","submitted_at":"2023-09-09T09:41:36Z","abstract_excerpt":"Multi-modal keyphrase generation aims to produce a set of keyphrases that represent the core points of the input text-image pair. In this regard, dominant methods mainly focus on multi-modal fusion for keyphrase generation. Nevertheless, there are still two main drawbacks: 1) only a limited number of sources, such as image captions, can be utilized to provide auxiliary information. However, they may not be sufficient for the subsequent keyphrase generation. 2) the input text and image are often not perfectly matched, and thus the image may introduce noise into the model. To address these limit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.04734","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.04734/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.04734","created_at":"2026-07-05T06:49:08.029203+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.04734v1","created_at":"2026-07-05T06:49:08.029203+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.04734","created_at":"2026-07-05T06:49:08.029203+00:00"},{"alias_kind":"pith_short_12","alias_value":"VRJL3XBMC5ZP","created_at":"2026-07-05T06:49:08.029203+00:00"},{"alias_kind":"pith_short_16","alias_value":"VRJL3XBMC5ZPYKHF","created_at":"2026-07-05T06:49:08.029203+00:00"},{"alias_kind":"pith_short_8","alias_value":"VRJL3XBM","created_at":"2026-07-05T06:49:08.029203+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VRJL3XBMC5ZPYKHFXGEC6ZKMFV","json":"https://pith.science/pith/VRJL3XBMC5ZPYKHFXGEC6ZKMFV.json","graph_json":"https://pith.science/api/pith-number/VRJL3XBMC5ZPYKHFXGEC6ZKMFV/graph.json","events_json":"https://pith.science/api/pith-number/VRJL3XBMC5ZPYKHFXGEC6ZKMFV/events.json","paper":"https://pith.science/paper/VRJL3XBM"},"agent_actions":{"view_html":"https://pith.science/pith/VRJL3XBMC5ZPYKHFXGEC6ZKMFV","download_json":"https://pith.science/pith/VRJL3XBMC5ZPYKHFXGEC6ZKMFV.json","view_paper":"https://pith.science/paper/VRJL3XBM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.04734&json=true","fetch_graph":"https://pith.science/api/pith-number/VRJL3XBMC5ZPYKHFXGEC6ZKMFV/graph.json","fetch_events":"https://pith.science/api/pith-number/VRJL3XBMC5ZPYKHFXGEC6ZKMFV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VRJL3XBMC5ZPYKHFXGEC6ZKMFV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VRJL3XBMC5ZPYKHFXGEC6ZKMFV/action/storage_attestation","attest_author":"https://pith.science/pith/VRJL3XBMC5ZPYKHFXGEC6ZKMFV/action/author_attestation","sign_citation":"https://pith.science/pith/VRJL3XBMC5ZPYKHFXGEC6ZKMFV/action/citation_signature","submit_replication":"https://pith.science/pith/VRJL3XBMC5ZPYKHFXGEC6ZKMFV/action/replication_record"}},"created_at":"2026-07-05T06:49:08.029203+00:00","updated_at":"2026-07-05T06:49:08.029203+00:00"}