{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BCLRSF5FZKJJ2SF3VZEPV2EUBJ","short_pith_number":"pith:BCLRSF5F","schema_version":"1.0","canonical_sha256":"08971917a5ca929d48bbae48fae8940a4248d1e45e4fef25dccee24d0d0566ca","source":{"kind":"arxiv","id":"2303.15616","version":1},"attestation_state":"computed","paper":{"title":"Fine-grained Audible Video Description","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aixuan Li, Bowen He, Dong Li, Jinxing Zhou, Lingpeng Kong, Meng Wang, Xiaodong Han, Xuyang Shen, Yiran Zhong, Yuchao Dai, Yu Qiao, Zhen Qin","submitted_at":"2023-03-27T22:03:48Z","abstract_excerpt":"We explore a new task for audio-visual-language modeling called fine-grained audible video description (FAVD). It aims to provide detailed textual descriptions for the given audible videos, including the appearance and spatial locations of each object, the actions of moving objects, and the sounds in videos. Existing visual-language modeling tasks often concentrate on visual cues in videos while undervaluing the language and audio modalities. On the other hand, FAVD requires not only audio-visual-language modeling skills but also paragraph-level language generation abilities. We construct the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.15616","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CV","submitted_at":"2023-03-27T22:03:48Z","cross_cats_sorted":[],"title_canon_sha256":"9c5c9506a7f1e4bbd2819de52ec719256b76bfa7725a597b2311da89a14322cf","abstract_canon_sha256":"c44165beb0b28ab617f1a8eaa3c62643c0d45b1ddb41a96aa1b8b51b8bc6e8cb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:56:42.475071Z","signature_b64":"y7Y4wR20F8H7rsJt3g21hxR3IDEx1on2yIIDkJCsupkXdxLB3FBQY5fAvvNyjpEDgq+qryBblkwYre2If6n8BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"08971917a5ca929d48bbae48fae8940a4248d1e45e4fef25dccee24d0d0566ca","last_reissued_at":"2026-07-05T05:56:42.474595Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:56:42.474595Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fine-grained Audible Video Description","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aixuan Li, Bowen He, Dong Li, Jinxing Zhou, Lingpeng Kong, Meng Wang, Xiaodong Han, Xuyang Shen, Yiran Zhong, Yuchao Dai, Yu Qiao, Zhen Qin","submitted_at":"2023-03-27T22:03:48Z","abstract_excerpt":"We explore a new task for audio-visual-language modeling called fine-grained audible video description (FAVD). It aims to provide detailed textual descriptions for the given audible videos, including the appearance and spatial locations of each object, the actions of moving objects, and the sounds in videos. Existing visual-language modeling tasks often concentrate on visual cues in videos while undervaluing the language and audio modalities. On the other hand, FAVD requires not only audio-visual-language modeling skills but also paragraph-level language generation abilities. We construct the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.15616","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.15616/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.15616","created_at":"2026-07-05T05:56:42.474653+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.15616v1","created_at":"2026-07-05T05:56:42.474653+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.15616","created_at":"2026-07-05T05:56:42.474653+00:00"},{"alias_kind":"pith_short_12","alias_value":"BCLRSF5FZKJJ","created_at":"2026-07-05T05:56:42.474653+00:00"},{"alias_kind":"pith_short_16","alias_value":"BCLRSF5FZKJJ2SF3","created_at":"2026-07-05T05:56:42.474653+00:00"},{"alias_kind":"pith_short_8","alias_value":"BCLRSF5F","created_at":"2026-07-05T05:56:42.474653+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BCLRSF5FZKJJ2SF3VZEPV2EUBJ","json":"https://pith.science/pith/BCLRSF5FZKJJ2SF3VZEPV2EUBJ.json","graph_json":"https://pith.science/api/pith-number/BCLRSF5FZKJJ2SF3VZEPV2EUBJ/graph.json","events_json":"https://pith.science/api/pith-number/BCLRSF5FZKJJ2SF3VZEPV2EUBJ/events.json","paper":"https://pith.science/paper/BCLRSF5F"},"agent_actions":{"view_html":"https://pith.science/pith/BCLRSF5FZKJJ2SF3VZEPV2EUBJ","download_json":"https://pith.science/pith/BCLRSF5FZKJJ2SF3VZEPV2EUBJ.json","view_paper":"https://pith.science/paper/BCLRSF5F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.15616&json=true","fetch_graph":"https://pith.science/api/pith-number/BCLRSF5FZKJJ2SF3VZEPV2EUBJ/graph.json","fetch_events":"https://pith.science/api/pith-number/BCLRSF5FZKJJ2SF3VZEPV2EUBJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BCLRSF5FZKJJ2SF3VZEPV2EUBJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BCLRSF5FZKJJ2SF3VZEPV2EUBJ/action/storage_attestation","attest_author":"https://pith.science/pith/BCLRSF5FZKJJ2SF3VZEPV2EUBJ/action/author_attestation","sign_citation":"https://pith.science/pith/BCLRSF5FZKJJ2SF3VZEPV2EUBJ/action/citation_signature","submit_replication":"https://pith.science/pith/BCLRSF5FZKJJ2SF3VZEPV2EUBJ/action/replication_record"}},"created_at":"2026-07-05T05:56:42.474653+00:00","updated_at":"2026-07-05T05:56:42.474653+00:00"}