{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:3OALKTCZLEG5ABLEM257UDZIGY","short_pith_number":"pith:3OALKTCZ","schema_version":"1.0","canonical_sha256":"db80b54c59590dd0056466bbfa0f28360e1f40aeca396161ea7a9be1b82ec279","source":{"kind":"arxiv","id":"2006.01595","version":1},"attestation_state":"computed","paper":{"title":"Large Scale Audiovisual Learning of Sounds with Weakly Labeled Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD","eess.IV","stat.ML"],"primary_cat":"eess.AS","authors_text":"Anurag Kumar, Haytham M. Fayek","submitted_at":"2020-05-29T01:30:14Z","abstract_excerpt":"Recognizing sounds is a key aspect of computational audio scene analysis and machine perception. In this paper, we advocate that sound recognition is inherently a multi-modal audiovisual task in that it is easier to differentiate sounds using both the audio and visual modalities as opposed to one or the other. We present an audiovisual fusion model that learns to recognize sounds from weakly labeled video recordings. The proposed fusion model utilizes an attention mechanism to dynamically combine the outputs of the individual audio and visual models. Experiments on the large scale sound events"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.01595","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2020-05-29T01:30:14Z","cross_cats_sorted":["cs.LG","cs.SD","eess.IV","stat.ML"],"title_canon_sha256":"82cbe085db0f2c65f8446bf3bb0956a98b12bf7b8dbcc9665114653ae3b2016c","abstract_canon_sha256":"30e221ac37b5335b16e26fd8578ae9ed2f51dc82918fb1cb7469b58b7c028521"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:07:19.158426Z","signature_b64":"ASoaNxXzDfJ76KQ6Jcf4BzKYeTVwegYDVMh6Uuo8biPALMgQxxMmXqStFiS2FfM2rsQW2AOT2c4B+cNNFI5aCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"db80b54c59590dd0056466bbfa0f28360e1f40aeca396161ea7a9be1b82ec279","last_reissued_at":"2026-07-05T01:07:19.157827Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:07:19.157827Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Scale Audiovisual Learning of Sounds with Weakly Labeled Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD","eess.IV","stat.ML"],"primary_cat":"eess.AS","authors_text":"Anurag Kumar, Haytham M. Fayek","submitted_at":"2020-05-29T01:30:14Z","abstract_excerpt":"Recognizing sounds is a key aspect of computational audio scene analysis and machine perception. In this paper, we advocate that sound recognition is inherently a multi-modal audiovisual task in that it is easier to differentiate sounds using both the audio and visual modalities as opposed to one or the other. We present an audiovisual fusion model that learns to recognize sounds from weakly labeled video recordings. The proposed fusion model utilizes an attention mechanism to dynamically combine the outputs of the individual audio and visual models. Experiments on the large scale sound events"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.01595","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.01595/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.01595","created_at":"2026-07-05T01:07:19.157952+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.01595v1","created_at":"2026-07-05T01:07:19.157952+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.01595","created_at":"2026-07-05T01:07:19.157952+00:00"},{"alias_kind":"pith_short_12","alias_value":"3OALKTCZLEG5","created_at":"2026-07-05T01:07:19.157952+00:00"},{"alias_kind":"pith_short_16","alias_value":"3OALKTCZLEG5ABLE","created_at":"2026-07-05T01:07:19.157952+00:00"},{"alias_kind":"pith_short_8","alias_value":"3OALKTCZ","created_at":"2026-07-05T01:07:19.157952+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.09323","citing_title":"Dynamic Inter-Class Confusion-Aware Encoder for Audio-Visual Fusion in Human Activity Recognition","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3OALKTCZLEG5ABLEM257UDZIGY","json":"https://pith.science/pith/3OALKTCZLEG5ABLEM257UDZIGY.json","graph_json":"https://pith.science/api/pith-number/3OALKTCZLEG5ABLEM257UDZIGY/graph.json","events_json":"https://pith.science/api/pith-number/3OALKTCZLEG5ABLEM257UDZIGY/events.json","paper":"https://pith.science/paper/3OALKTCZ"},"agent_actions":{"view_html":"https://pith.science/pith/3OALKTCZLEG5ABLEM257UDZIGY","download_json":"https://pith.science/pith/3OALKTCZLEG5ABLEM257UDZIGY.json","view_paper":"https://pith.science/paper/3OALKTCZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.01595&json=true","fetch_graph":"https://pith.science/api/pith-number/3OALKTCZLEG5ABLEM257UDZIGY/graph.json","fetch_events":"https://pith.science/api/pith-number/3OALKTCZLEG5ABLEM257UDZIGY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3OALKTCZLEG5ABLEM257UDZIGY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3OALKTCZLEG5ABLEM257UDZIGY/action/storage_attestation","attest_author":"https://pith.science/pith/3OALKTCZLEG5ABLEM257UDZIGY/action/author_attestation","sign_citation":"https://pith.science/pith/3OALKTCZLEG5ABLEM257UDZIGY/action/citation_signature","submit_replication":"https://pith.science/pith/3OALKTCZLEG5ABLEM257UDZIGY/action/replication_record"}},"created_at":"2026-07-05T01:07:19.157952+00:00","updated_at":"2026-07-05T01:07:19.157952+00:00"}