{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:KPEYQSLQKQSSIJV3MHZY4KYF5W","short_pith_number":"pith:KPEYQSLQ","schema_version":"1.0","canonical_sha256":"53c988497054252426bb61f38e2b05ed97646e48b4f72dc5b6f77f7c76b9c54f","source":{"kind":"arxiv","id":"2203.06356","version":1},"attestation_state":"computed","paper":{"title":"Taking an Emotional Look at Video Paragraph Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang Wen Chen, Hanli Wang, Qinyu Li, Tengpeng Li","submitted_at":"2022-03-12T06:19:48Z","abstract_excerpt":"Translating visual data into natural language is essential for machines to understand the world and interact with humans. In this work, a comprehensive study is conducted on video paragraph captioning, with the goal to generate paragraph-level descriptions for a given video. However, current researches mainly focus on detecting objective facts, ignoring the needs to establish the logical associations between sentences and to discover more accurate emotions related to video contents. Such a problem impairs fluent and abundant expressions of predicted captions, which are far below human language"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.06356","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-03-12T06:19:48Z","cross_cats_sorted":[],"title_canon_sha256":"1d199ff315a3f8820ca2f73ad5a46268667d9c462a413f0706dac6ebd8f8c7e0","abstract_canon_sha256":"9e93815181d1147bfc1f20acc3e467a192b88b31bd71216a981d4572d43b665e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:04:49.925472Z","signature_b64":"ZMvXmQNnfiInZKMJLUWiA5+z7RZK5lszCn9Wkbk6Ygwv3nolDJJ6o0RoxdeiZnwK3NfJDR9tCbTP6FMWk1kuCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"53c988497054252426bb61f38e2b05ed97646e48b4f72dc5b6f77f7c76b9c54f","last_reissued_at":"2026-07-05T04:04:49.925098Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:04:49.925098Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Taking an Emotional Look at Video Paragraph Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang Wen Chen, Hanli Wang, Qinyu Li, Tengpeng Li","submitted_at":"2022-03-12T06:19:48Z","abstract_excerpt":"Translating visual data into natural language is essential for machines to understand the world and interact with humans. In this work, a comprehensive study is conducted on video paragraph captioning, with the goal to generate paragraph-level descriptions for a given video. However, current researches mainly focus on detecting objective facts, ignoring the needs to establish the logical associations between sentences and to discover more accurate emotions related to video contents. Such a problem impairs fluent and abundant expressions of predicted captions, which are far below human language"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.06356","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.06356/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.06356","created_at":"2026-07-05T04:04:49.925152+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.06356v1","created_at":"2026-07-05T04:04:49.925152+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.06356","created_at":"2026-07-05T04:04:49.925152+00:00"},{"alias_kind":"pith_short_12","alias_value":"KPEYQSLQKQSS","created_at":"2026-07-05T04:04:49.925152+00:00"},{"alias_kind":"pith_short_16","alias_value":"KPEYQSLQKQSSIJV3","created_at":"2026-07-05T04:04:49.925152+00:00"},{"alias_kind":"pith_short_8","alias_value":"KPEYQSLQ","created_at":"2026-07-05T04:04:49.925152+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.08861","citing_title":"Generative Planning with 3D-vision Language Pre-training for End-to-End Autonomous Driving","ref_index":25,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KPEYQSLQKQSSIJV3MHZY4KYF5W","json":"https://pith.science/pith/KPEYQSLQKQSSIJV3MHZY4KYF5W.json","graph_json":"https://pith.science/api/pith-number/KPEYQSLQKQSSIJV3MHZY4KYF5W/graph.json","events_json":"https://pith.science/api/pith-number/KPEYQSLQKQSSIJV3MHZY4KYF5W/events.json","paper":"https://pith.science/paper/KPEYQSLQ"},"agent_actions":{"view_html":"https://pith.science/pith/KPEYQSLQKQSSIJV3MHZY4KYF5W","download_json":"https://pith.science/pith/KPEYQSLQKQSSIJV3MHZY4KYF5W.json","view_paper":"https://pith.science/paper/KPEYQSLQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.06356&json=true","fetch_graph":"https://pith.science/api/pith-number/KPEYQSLQKQSSIJV3MHZY4KYF5W/graph.json","fetch_events":"https://pith.science/api/pith-number/KPEYQSLQKQSSIJV3MHZY4KYF5W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KPEYQSLQKQSSIJV3MHZY4KYF5W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KPEYQSLQKQSSIJV3MHZY4KYF5W/action/storage_attestation","attest_author":"https://pith.science/pith/KPEYQSLQKQSSIJV3MHZY4KYF5W/action/author_attestation","sign_citation":"https://pith.science/pith/KPEYQSLQKQSSIJV3MHZY4KYF5W/action/citation_signature","submit_replication":"https://pith.science/pith/KPEYQSLQKQSSIJV3MHZY4KYF5W/action/replication_record"}},"created_at":"2026-07-05T04:04:49.925152+00:00","updated_at":"2026-07-05T04:04:49.925152+00:00"}