{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UHXQFTPG43T2GEVK5TZXWGTRDP","short_pith_number":"pith:UHXQFTPG","schema_version":"1.0","canonical_sha256":"a1ef02cde6e6e7a312aaecf37b1a711bde342d1c6f3c27e7f1ba9835f0c35ac4","source":{"kind":"arxiv","id":"2501.09291","version":2},"attestation_state":"computed","paper":{"title":"LAVCap: LLM-based Audio-Visual Captioning using Optimal Transport","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Hyeongkeun Lee, Joon Son Chung, Kyeongha Rho, Valentio Iverson","submitted_at":"2025-01-16T04:53:29Z","abstract_excerpt":"Automated audio captioning is a task that generates textual descriptions for audio content, and recent studies have explored using visual information to enhance captioning quality. However, current methods often fail to effectively fuse audio and visual data, missing important semantic cues from each modality. To address this, we introduce LAVCap, a large language model (LLM)-based audio-visual captioning framework that effectively integrates visual information with audio to improve audio captioning performance. LAVCap employs an optimal transport-based alignment loss to bridge the modality ga"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.09291","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MM","submitted_at":"2025-01-16T04:53:29Z","cross_cats_sorted":["cs.AI","cs.SD","eess.AS"],"title_canon_sha256":"2af0d1d6fec176a1b83740215c720f08c728219e735c4a5ad77490ccaa84fd10","abstract_canon_sha256":"96d71b05421a0d706df88268683f705bbefd7f9c8029b539ce1f96ede929e662"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:53.174765Z","signature_b64":"F/4h/gHNlMLGBRWMLAO73SbLib95gggUrTx3o1BYoKrQJIfnielFwG3Js0bCi82FkdlDMOD6tnxk3EXbn7d3Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a1ef02cde6e6e7a312aaecf37b1a711bde342d1c6f3c27e7f1ba9835f0c35ac4","last_reissued_at":"2026-07-05T10:31:53.174308Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:53.174308Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LAVCap: LLM-based Audio-Visual Captioning using Optimal Transport","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Hyeongkeun Lee, Joon Son Chung, Kyeongha Rho, Valentio Iverson","submitted_at":"2025-01-16T04:53:29Z","abstract_excerpt":"Automated audio captioning is a task that generates textual descriptions for audio content, and recent studies have explored using visual information to enhance captioning quality. However, current methods often fail to effectively fuse audio and visual data, missing important semantic cues from each modality. To address this, we introduce LAVCap, a large language model (LLM)-based audio-visual captioning framework that effectively integrates visual information with audio to improve audio captioning performance. LAVCap employs an optimal transport-based alignment loss to bridge the modality ga"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.09291","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.09291/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.09291","created_at":"2026-07-05T10:31:53.174366+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.09291v2","created_at":"2026-07-05T10:31:53.174366+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.09291","created_at":"2026-07-05T10:31:53.174366+00:00"},{"alias_kind":"pith_short_12","alias_value":"UHXQFTPG43T2","created_at":"2026-07-05T10:31:53.174366+00:00"},{"alias_kind":"pith_short_16","alias_value":"UHXQFTPG43T2GEVK","created_at":"2026-07-05T10:31:53.174366+00:00"},{"alias_kind":"pith_short_8","alias_value":"UHXQFTPG","created_at":"2026-07-05T10:31:53.174366+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.05899","citing_title":"WhisQ: Cross-Modal Representation Learning for Text-to-Music MOS Prediction","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UHXQFTPG43T2GEVK5TZXWGTRDP","json":"https://pith.science/pith/UHXQFTPG43T2GEVK5TZXWGTRDP.json","graph_json":"https://pith.science/api/pith-number/UHXQFTPG43T2GEVK5TZXWGTRDP/graph.json","events_json":"https://pith.science/api/pith-number/UHXQFTPG43T2GEVK5TZXWGTRDP/events.json","paper":"https://pith.science/paper/UHXQFTPG"},"agent_actions":{"view_html":"https://pith.science/pith/UHXQFTPG43T2GEVK5TZXWGTRDP","download_json":"https://pith.science/pith/UHXQFTPG43T2GEVK5TZXWGTRDP.json","view_paper":"https://pith.science/paper/UHXQFTPG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.09291&json=true","fetch_graph":"https://pith.science/api/pith-number/UHXQFTPG43T2GEVK5TZXWGTRDP/graph.json","fetch_events":"https://pith.science/api/pith-number/UHXQFTPG43T2GEVK5TZXWGTRDP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UHXQFTPG43T2GEVK5TZXWGTRDP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UHXQFTPG43T2GEVK5TZXWGTRDP/action/storage_attestation","attest_author":"https://pith.science/pith/UHXQFTPG43T2GEVK5TZXWGTRDP/action/author_attestation","sign_citation":"https://pith.science/pith/UHXQFTPG43T2GEVK5TZXWGTRDP/action/citation_signature","submit_replication":"https://pith.science/pith/UHXQFTPG43T2GEVK5TZXWGTRDP/action/replication_record"}},"created_at":"2026-07-05T10:31:53.174366+00:00","updated_at":"2026-07-05T10:31:53.174366+00:00"}