{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:LSR4PQC6BH372YW2CKYPE7EDUS","short_pith_number":"pith:LSR4PQC6","schema_version":"1.0","canonical_sha256":"5ca3c7c05e09f7fd62da12b0f27c83a4a64f344f56f3c5b380761d2aa295fb04","source":{"kind":"arxiv","id":"1911.03821","version":2},"attestation_state":"computed","paper":{"title":"Adaptive Fusion Techniques for Multimodal Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG","eess.AS"],"primary_cat":"cs.CL","authors_text":"Gaurav Sahu, Olga Vechtomova","submitted_at":"2019-11-10T01:39:46Z","abstract_excerpt":"Effective fusion of data from multiple modalities, such as video, speech, and text, is challenging due to the heterogeneous nature of multimodal data. In this paper, we propose adaptive fusion techniques that aim to model context from different modalities effectively. Instead of defining a deterministic fusion operation, such as concatenation, for the network, we let the network decide \"how\" to combine a given set of multimodal features more effectively. We propose two networks: 1) Auto-Fusion, which learns to compress information from different modalities while preserving the context, and 2) "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1911.03821","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-11-10T01:39:46Z","cross_cats_sorted":["cs.CV","cs.LG","eess.AS"],"title_canon_sha256":"b672450297a21065e20a1824b40910d739a348520778f42fb5230ac1c3948425","abstract_canon_sha256":"df6826c414f35611fd56b8ac0593de2773482f6b21f2a6a1af5c9f91441c5353"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:09:35.754105Z","signature_b64":"TgmBTa2w5DZFpSLY8w3i++GU5TfxZxRH33ofRbGU/SZdHZRAOLHcBRapQ4Jf7NqgElIVt9zheovOYZJOtxoyAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ca3c7c05e09f7fd62da12b0f27c83a4a64f344f56f3c5b380761d2aa295fb04","last_reissued_at":"2026-07-05T02:09:35.753762Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:09:35.753762Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Fusion Techniques for Multimodal Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG","eess.AS"],"primary_cat":"cs.CL","authors_text":"Gaurav Sahu, Olga Vechtomova","submitted_at":"2019-11-10T01:39:46Z","abstract_excerpt":"Effective fusion of data from multiple modalities, such as video, speech, and text, is challenging due to the heterogeneous nature of multimodal data. In this paper, we propose adaptive fusion techniques that aim to model context from different modalities effectively. Instead of defining a deterministic fusion operation, such as concatenation, for the network, we let the network decide \"how\" to combine a given set of multimodal features more effectively. We propose two networks: 1) Auto-Fusion, which learns to compress information from different modalities while preserving the context, and 2) "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1911.03821","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1911.03821/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1911.03821","created_at":"2026-07-05T02:09:35.753820+00:00"},{"alias_kind":"arxiv_version","alias_value":"1911.03821v2","created_at":"2026-07-05T02:09:35.753820+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.03821","created_at":"2026-07-05T02:09:35.753820+00:00"},{"alias_kind":"pith_short_12","alias_value":"LSR4PQC6BH37","created_at":"2026-07-05T02:09:35.753820+00:00"},{"alias_kind":"pith_short_16","alias_value":"LSR4PQC6BH372YW2","created_at":"2026-07-05T02:09:35.753820+00:00"},{"alias_kind":"pith_short_8","alias_value":"LSR4PQC6","created_at":"2026-07-05T02:09:35.753820+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.22676","citing_title":"Listening to the Unspoken: Exploring \"365\" Aspects of Multimodal Interview Performance Assessment","ref_index":36,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LSR4PQC6BH372YW2CKYPE7EDUS","json":"https://pith.science/pith/LSR4PQC6BH372YW2CKYPE7EDUS.json","graph_json":"https://pith.science/api/pith-number/LSR4PQC6BH372YW2CKYPE7EDUS/graph.json","events_json":"https://pith.science/api/pith-number/LSR4PQC6BH372YW2CKYPE7EDUS/events.json","paper":"https://pith.science/paper/LSR4PQC6"},"agent_actions":{"view_html":"https://pith.science/pith/LSR4PQC6BH372YW2CKYPE7EDUS","download_json":"https://pith.science/pith/LSR4PQC6BH372YW2CKYPE7EDUS.json","view_paper":"https://pith.science/paper/LSR4PQC6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1911.03821&json=true","fetch_graph":"https://pith.science/api/pith-number/LSR4PQC6BH372YW2CKYPE7EDUS/graph.json","fetch_events":"https://pith.science/api/pith-number/LSR4PQC6BH372YW2CKYPE7EDUS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LSR4PQC6BH372YW2CKYPE7EDUS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LSR4PQC6BH372YW2CKYPE7EDUS/action/storage_attestation","attest_author":"https://pith.science/pith/LSR4PQC6BH372YW2CKYPE7EDUS/action/author_attestation","sign_citation":"https://pith.science/pith/LSR4PQC6BH372YW2CKYPE7EDUS/action/citation_signature","submit_replication":"https://pith.science/pith/LSR4PQC6BH372YW2CKYPE7EDUS/action/replication_record"}},"created_at":"2026-07-05T02:09:35.753820+00:00","updated_at":"2026-07-05T02:09:35.753820+00:00"}