{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:COP266PWK5X34BI6HSQHVXKPQK","short_pith_number":"pith:COP266PW","schema_version":"1.0","canonical_sha256":"139faf79f6576fbe051e3ca07add4f82ae72cee160b973c2c6c12aa1fd56f5df","source":{"kind":"arxiv","id":"2406.06499","version":3},"attestation_state":"computed","paper":{"title":"NarrativeBridge: Enhancing Video Captioning with Causal-Temporal Narrative","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CV","authors_text":"Adrian Hilton, Armin Mustafa, Asmar Nadeem, Faegheh Sardari, Robert Dawes, Syed Sameed Husain","submitted_at":"2024-06-10T17:34:24Z","abstract_excerpt":"Existing video captioning benchmarks and models lack causal-temporal narrative, which is sequences of events linked through cause and effect, unfolding over time and driven by characters or agents. This lack of narrative restricts models' ability to generate text descriptions that capture the causal and temporal dynamics inherent in video content. To address this gap, we propose NarrativeBridge, an approach comprising of: (1) a novel Causal-Temporal Narrative (CTN) captions benchmark generated using a large language model and few-shot prompting, explicitly encoding cause-effect temporal relati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.06499","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-10T17:34:24Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"a4d6446e10843b31b48f66b2da5c33bb474909ecc21c825a31aa9a0347a238c8","abstract_canon_sha256":"c1758dbbcf41a45a5fa99024ad1545ffd3e886245941938da9f17b663a017c6a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:41.546590Z","signature_b64":"ntIEOKhqpEL5DIPBiYeXur/60v2xxlwMua6/FBoILtHjhenGLwEGUONzaRE1+3XCk8q7o28zVeM3sG6b87pQAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"139faf79f6576fbe051e3ca07add4f82ae72cee160b973c2c6c12aa1fd56f5df","last_reissued_at":"2026-07-05T10:14:41.546066Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:41.546066Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NarrativeBridge: Enhancing Video Captioning with Causal-Temporal Narrative","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CV","authors_text":"Adrian Hilton, Armin Mustafa, Asmar Nadeem, Faegheh Sardari, Robert Dawes, Syed Sameed Husain","submitted_at":"2024-06-10T17:34:24Z","abstract_excerpt":"Existing video captioning benchmarks and models lack causal-temporal narrative, which is sequences of events linked through cause and effect, unfolding over time and driven by characters or agents. This lack of narrative restricts models' ability to generate text descriptions that capture the causal and temporal dynamics inherent in video content. To address this gap, we propose NarrativeBridge, an approach comprising of: (1) a novel Causal-Temporal Narrative (CTN) captions benchmark generated using a large language model and few-shot prompting, explicitly encoding cause-effect temporal relati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.06499","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.06499/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.06499","created_at":"2026-07-05T10:14:41.546128+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.06499v3","created_at":"2026-07-05T10:14:41.546128+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.06499","created_at":"2026-07-05T10:14:41.546128+00:00"},{"alias_kind":"pith_short_12","alias_value":"COP266PWK5X3","created_at":"2026-07-05T10:14:41.546128+00:00"},{"alias_kind":"pith_short_16","alias_value":"COP266PWK5X34BI6","created_at":"2026-07-05T10:14:41.546128+00:00"},{"alias_kind":"pith_short_8","alias_value":"COP266PW","created_at":"2026-07-05T10:14:41.546128+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04815","citing_title":"From Vision To Language through Graph of Events in Space and Time: An Explainable Self-supervised Approach","ref_index":55,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/COP266PWK5X34BI6HSQHVXKPQK","json":"https://pith.science/pith/COP266PWK5X34BI6HSQHVXKPQK.json","graph_json":"https://pith.science/api/pith-number/COP266PWK5X34BI6HSQHVXKPQK/graph.json","events_json":"https://pith.science/api/pith-number/COP266PWK5X34BI6HSQHVXKPQK/events.json","paper":"https://pith.science/paper/COP266PW"},"agent_actions":{"view_html":"https://pith.science/pith/COP266PWK5X34BI6HSQHVXKPQK","download_json":"https://pith.science/pith/COP266PWK5X34BI6HSQHVXKPQK.json","view_paper":"https://pith.science/paper/COP266PW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.06499&json=true","fetch_graph":"https://pith.science/api/pith-number/COP266PWK5X34BI6HSQHVXKPQK/graph.json","fetch_events":"https://pith.science/api/pith-number/COP266PWK5X34BI6HSQHVXKPQK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/COP266PWK5X34BI6HSQHVXKPQK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/COP266PWK5X34BI6HSQHVXKPQK/action/storage_attestation","attest_author":"https://pith.science/pith/COP266PWK5X34BI6HSQHVXKPQK/action/author_attestation","sign_citation":"https://pith.science/pith/COP266PWK5X34BI6HSQHVXKPQK/action/citation_signature","submit_replication":"https://pith.science/pith/COP266PWK5X34BI6HSQHVXKPQK/action/replication_record"}},"created_at":"2026-07-05T10:14:41.546128+00:00","updated_at":"2026-07-05T10:14:41.546128+00:00"}