{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:CBUCJXSY2NR5XPJLUQPRI5UWAE","short_pith_number":"pith:CBUCJXSY","schema_version":"1.0","canonical_sha256":"106824de58d363dbbd2ba41f147696013bc815f8fc675f0880bd89650037bc5d","source":{"kind":"arxiv","id":"1911.12667","version":3},"attestation_state":"computed","paper":{"title":"Self-Supervised Learning by Cross-Modal Audio-Video Clustering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bernard Ghanem, Bruno Korbar, Dhruv Mahajan, Du Tran, Humam Alwassel, Lorenzo Torresani","submitted_at":"2019-11-28T12:17:36Z","abstract_excerpt":"Visual and audio modalities are highly correlated, yet they contain different information. Their strong correlation makes it possible to predict the semantics of one from the other with good accuracy. Their intrinsic differences make cross-modal prediction a potentially more rewarding pretext task for self-supervised learning of video and audio representations compared to within-modality learning. Based on this intuition, we propose Cross-Modal Deep Clustering (XDC), a novel self-supervised method that leverages unsupervised clustering in one modality (e.g., audio) as a supervisory signal for "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1911.12667","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-11-28T12:17:36Z","cross_cats_sorted":[],"title_canon_sha256":"80856970122999432efc431e06677168b8337dafa243f571dc66d33ba1861f81","abstract_canon_sha256":"54fea15425a2698b738e1e58251ce9324a4f1cc50ca24f141618c1ec9968a16f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:45:59.598128Z","signature_b64":"OB0wxusIcOS+xnQZmiDa1JFYug1T1hZxh7rrgAw/xooKJMW+hqOqgYzEypXrjGGHadjQuoUXhybfE68LfV3ZDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"106824de58d363dbbd2ba41f147696013bc815f8fc675f0880bd89650037bc5d","last_reissued_at":"2026-07-05T01:45:59.597623Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:45:59.597623Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Supervised Learning by Cross-Modal Audio-Video Clustering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bernard Ghanem, Bruno Korbar, Dhruv Mahajan, Du Tran, Humam Alwassel, Lorenzo Torresani","submitted_at":"2019-11-28T12:17:36Z","abstract_excerpt":"Visual and audio modalities are highly correlated, yet they contain different information. Their strong correlation makes it possible to predict the semantics of one from the other with good accuracy. Their intrinsic differences make cross-modal prediction a potentially more rewarding pretext task for self-supervised learning of video and audio representations compared to within-modality learning. Based on this intuition, we propose Cross-Modal Deep Clustering (XDC), a novel self-supervised method that leverages unsupervised clustering in one modality (e.g., audio) as a supervisory signal for "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1911.12667","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1911.12667/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1911.12667","created_at":"2026-07-05T01:45:59.597683+00:00"},{"alias_kind":"arxiv_version","alias_value":"1911.12667v3","created_at":"2026-07-05T01:45:59.597683+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.12667","created_at":"2026-07-05T01:45:59.597683+00:00"},{"alias_kind":"pith_short_12","alias_value":"CBUCJXSY2NR5","created_at":"2026-07-05T01:45:59.597683+00:00"},{"alias_kind":"pith_short_16","alias_value":"CBUCJXSY2NR5XPJL","created_at":"2026-07-05T01:45:59.597683+00:00"},{"alias_kind":"pith_short_8","alias_value":"CBUCJXSY","created_at":"2026-07-05T01:45:59.597683+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16403","citing_title":"When Vision Speaks for Sound","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CBUCJXSY2NR5XPJLUQPRI5UWAE","json":"https://pith.science/pith/CBUCJXSY2NR5XPJLUQPRI5UWAE.json","graph_json":"https://pith.science/api/pith-number/CBUCJXSY2NR5XPJLUQPRI5UWAE/graph.json","events_json":"https://pith.science/api/pith-number/CBUCJXSY2NR5XPJLUQPRI5UWAE/events.json","paper":"https://pith.science/paper/CBUCJXSY"},"agent_actions":{"view_html":"https://pith.science/pith/CBUCJXSY2NR5XPJLUQPRI5UWAE","download_json":"https://pith.science/pith/CBUCJXSY2NR5XPJLUQPRI5UWAE.json","view_paper":"https://pith.science/paper/CBUCJXSY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1911.12667&json=true","fetch_graph":"https://pith.science/api/pith-number/CBUCJXSY2NR5XPJLUQPRI5UWAE/graph.json","fetch_events":"https://pith.science/api/pith-number/CBUCJXSY2NR5XPJLUQPRI5UWAE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CBUCJXSY2NR5XPJLUQPRI5UWAE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CBUCJXSY2NR5XPJLUQPRI5UWAE/action/storage_attestation","attest_author":"https://pith.science/pith/CBUCJXSY2NR5XPJLUQPRI5UWAE/action/author_attestation","sign_citation":"https://pith.science/pith/CBUCJXSY2NR5XPJLUQPRI5UWAE/action/citation_signature","submit_replication":"https://pith.science/pith/CBUCJXSY2NR5XPJLUQPRI5UWAE/action/replication_record"}},"created_at":"2026-07-05T01:45:59.597683+00:00","updated_at":"2026-07-05T01:45:59.597683+00:00"}