{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:LL2ASMUS2B37SMYA7ATOLB6D7S","short_pith_number":"pith:LL2ASMUS","schema_version":"1.0","canonical_sha256":"5af4093292d077f93300f826e587c3fc9934a0d53861494ec6b44ee7f3caa1f8","source":{"kind":"arxiv","id":"1908.02127","version":1},"attestation_state":"computed","paper":{"title":"Aligning Linguistic Words and Visual Semantic Units for Image Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Hanqing Lu, Jiangwei Li, Jing Liu, Jinhui Tang, Longteng Guo, Wei Luo","submitted_at":"2019-08-06T13:19:24Z","abstract_excerpt":"Image captioning attempts to generate a sentence composed of several linguistic words, which are used to describe objects, attributes, and interactions in an image, denoted as visual semantic units in this paper. Based on this view, we propose to explicitly model the object interactions in semantics and geometry based on Graph Convolutional Networks (GCNs), and fully exploit the alignment between linguistic words and visual semantic units for image captioning. Particularly, we construct a semantic graph and a geometry graph, where each node corresponds to a visual semantic unit, i.e., an objec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.02127","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-08-06T13:19:24Z","cross_cats_sorted":["cs.CL","cs.LG","cs.MM"],"title_canon_sha256":"3e6427a119632c6ec44b6a0148daa56d520b371371cb3abfb0be0659082737e9","abstract_canon_sha256":"24e854e5156979b345fbaf1ca1abb93356f51f9104a6fdd998121d5db5f64ee4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T23:51:51.413595Z","signature_b64":"uWQ8kCK2NdUfGek2w19cHmMtGO0RcuHR6cFgFNn3BN8JG10dS+2GBO7xn1ShfNPNu+K/rwHm8MoMcoNDDBddDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5af4093292d077f93300f826e587c3fc9934a0d53861494ec6b44ee7f3caa1f8","last_reissued_at":"2026-07-04T23:51:51.413192Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T23:51:51.413192Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Aligning Linguistic Words and Visual Semantic Units for Image Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Hanqing Lu, Jiangwei Li, Jing Liu, Jinhui Tang, Longteng Guo, Wei Luo","submitted_at":"2019-08-06T13:19:24Z","abstract_excerpt":"Image captioning attempts to generate a sentence composed of several linguistic words, which are used to describe objects, attributes, and interactions in an image, denoted as visual semantic units in this paper. Based on this view, we propose to explicitly model the object interactions in semantics and geometry based on Graph Convolutional Networks (GCNs), and fully exploit the alignment between linguistic words and visual semantic units for image captioning. Particularly, we construct a semantic graph and a geometry graph, where each node corresponds to a visual semantic unit, i.e., an objec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.02127","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.02127/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.02127","created_at":"2026-07-04T23:51:51.413260+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.02127v1","created_at":"2026-07-04T23:51:51.413260+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.02127","created_at":"2026-07-04T23:51:51.413260+00:00"},{"alias_kind":"pith_short_12","alias_value":"LL2ASMUS2B37","created_at":"2026-07-04T23:51:51.413260+00:00"},{"alias_kind":"pith_short_16","alias_value":"LL2ASMUS2B37SMYA","created_at":"2026-07-04T23:51:51.413260+00:00"},{"alias_kind":"pith_short_8","alias_value":"LL2ASMUS","created_at":"2026-07-04T23:51:51.413260+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LL2ASMUS2B37SMYA7ATOLB6D7S","json":"https://pith.science/pith/LL2ASMUS2B37SMYA7ATOLB6D7S.json","graph_json":"https://pith.science/api/pith-number/LL2ASMUS2B37SMYA7ATOLB6D7S/graph.json","events_json":"https://pith.science/api/pith-number/LL2ASMUS2B37SMYA7ATOLB6D7S/events.json","paper":"https://pith.science/paper/LL2ASMUS"},"agent_actions":{"view_html":"https://pith.science/pith/LL2ASMUS2B37SMYA7ATOLB6D7S","download_json":"https://pith.science/pith/LL2ASMUS2B37SMYA7ATOLB6D7S.json","view_paper":"https://pith.science/paper/LL2ASMUS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.02127&json=true","fetch_graph":"https://pith.science/api/pith-number/LL2ASMUS2B37SMYA7ATOLB6D7S/graph.json","fetch_events":"https://pith.science/api/pith-number/LL2ASMUS2B37SMYA7ATOLB6D7S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LL2ASMUS2B37SMYA7ATOLB6D7S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LL2ASMUS2B37SMYA7ATOLB6D7S/action/storage_attestation","attest_author":"https://pith.science/pith/LL2ASMUS2B37SMYA7ATOLB6D7S/action/author_attestation","sign_citation":"https://pith.science/pith/LL2ASMUS2B37SMYA7ATOLB6D7S/action/citation_signature","submit_replication":"https://pith.science/pith/LL2ASMUS2B37SMYA7ATOLB6D7S/action/replication_record"}},"created_at":"2026-07-04T23:51:51.413260+00:00","updated_at":"2026-07-04T23:51:51.413260+00:00"}