{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NW4TSDIYKVSONY6T7IRJPFHPA5","short_pith_number":"pith:NW4TSDIY","schema_version":"1.0","canonical_sha256":"6db9390d185564e6e3d3fa229794ef07666de2827b60986b99a62190973e332b","source":{"kind":"arxiv","id":"2308.06685","version":1},"attestation_state":"computed","paper":{"title":"Video Captioning with Aggregated Features Based on Dual Graphs and Gated Fusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bin Liu, Jing Wang, Yutao Jin","submitted_at":"2023-08-13T05:18:08Z","abstract_excerpt":"The application of video captioning models aims at translating the content of videos by using accurate natural language. Due to the complex nature inbetween object interaction in the video, the comprehensive understanding of spatio-temporal relations of objects remains a challenging task. Existing methods often fail in generating sufficient feature representations of video content. In this paper, we propose a video captioning model based on dual graphs and gated fusion: we adapt two types of graphs to generate feature representations of video content and utilize gated fusion to further underst"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.06685","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-08-13T05:18:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"805f8d39e25369f4acfa7b2081abc217b6e9f748863e27b3e35f0665db3b2a9f","abstract_canon_sha256":"4de678feab3afa5860ca5aded7725a6babb1237b7d5b29ce1fdb48b48faa895b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:40:43.767251Z","signature_b64":"JfOR/eYe+rDjmzOZQI1r171vjrGWkc11Mc7AN1zGLBQQm+sDC1qsAdoSN1vj5SH890/boz86Xu5cVOsxK5TlCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6db9390d185564e6e3d3fa229794ef07666de2827b60986b99a62190973e332b","last_reissued_at":"2026-07-05T06:40:43.766807Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:40:43.766807Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video Captioning with Aggregated Features Based on Dual Graphs and Gated Fusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bin Liu, Jing Wang, Yutao Jin","submitted_at":"2023-08-13T05:18:08Z","abstract_excerpt":"The application of video captioning models aims at translating the content of videos by using accurate natural language. Due to the complex nature inbetween object interaction in the video, the comprehensive understanding of spatio-temporal relations of objects remains a challenging task. Existing methods often fail in generating sufficient feature representations of video content. In this paper, we propose a video captioning model based on dual graphs and gated fusion: we adapt two types of graphs to generate feature representations of video content and utilize gated fusion to further underst"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.06685","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.06685/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.06685","created_at":"2026-07-05T06:40:43.766873+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.06685v1","created_at":"2026-07-05T06:40:43.766873+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.06685","created_at":"2026-07-05T06:40:43.766873+00:00"},{"alias_kind":"pith_short_12","alias_value":"NW4TSDIYKVSO","created_at":"2026-07-05T06:40:43.766873+00:00"},{"alias_kind":"pith_short_16","alias_value":"NW4TSDIYKVSONY6T","created_at":"2026-07-05T06:40:43.766873+00:00"},{"alias_kind":"pith_short_8","alias_value":"NW4TSDIY","created_at":"2026-07-05T06:40:43.766873+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.10720","citing_title":"Bridging Vision and Language: Modeling Causality and Temporality in Video Narratives","ref_index":2023,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NW4TSDIYKVSONY6T7IRJPFHPA5","json":"https://pith.science/pith/NW4TSDIYKVSONY6T7IRJPFHPA5.json","graph_json":"https://pith.science/api/pith-number/NW4TSDIYKVSONY6T7IRJPFHPA5/graph.json","events_json":"https://pith.science/api/pith-number/NW4TSDIYKVSONY6T7IRJPFHPA5/events.json","paper":"https://pith.science/paper/NW4TSDIY"},"agent_actions":{"view_html":"https://pith.science/pith/NW4TSDIYKVSONY6T7IRJPFHPA5","download_json":"https://pith.science/pith/NW4TSDIYKVSONY6T7IRJPFHPA5.json","view_paper":"https://pith.science/paper/NW4TSDIY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.06685&json=true","fetch_graph":"https://pith.science/api/pith-number/NW4TSDIYKVSONY6T7IRJPFHPA5/graph.json","fetch_events":"https://pith.science/api/pith-number/NW4TSDIYKVSONY6T7IRJPFHPA5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NW4TSDIYKVSONY6T7IRJPFHPA5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NW4TSDIYKVSONY6T7IRJPFHPA5/action/storage_attestation","attest_author":"https://pith.science/pith/NW4TSDIYKVSONY6T7IRJPFHPA5/action/author_attestation","sign_citation":"https://pith.science/pith/NW4TSDIYKVSONY6T7IRJPFHPA5/action/citation_signature","submit_replication":"https://pith.science/pith/NW4TSDIYKVSONY6T7IRJPFHPA5/action/replication_record"}},"created_at":"2026-07-05T06:40:43.766873+00:00","updated_at":"2026-07-05T06:40:43.766873+00:00"}