{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZLHXYK6LISPF7RV4UPOYEAXERM","short_pith_number":"pith:ZLHXYK6L","schema_version":"1.0","canonical_sha256":"cacf7c2bcb449e5fc6bca3dd8202e48b178dd2a2ac4b4792b8cc916250398559","source":{"kind":"arxiv","id":"2403.19638","version":1},"attestation_state":"computed","paper":{"title":"Siamese Vision Transformers are Scalable Audio-visual Learners","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Gedas Bertasius, Yan-Bo Lin","submitted_at":"2024-03-28T17:52:24Z","abstract_excerpt":"Traditional audio-visual methods rely on independent audio and visual backbones, which is costly and not scalable. In this work, we investigate using an audio-visual siamese network (AVSiam) for efficient and scalable audio-visual pretraining. Our framework uses a single shared vision transformer backbone to process audio and visual inputs, improving its parameter efficiency, reducing the GPU memory footprint, and allowing us to scale our method to larger datasets and model sizes. We pretrain our model using a contrastive audio-visual matching objective with a multi-ratio random masking scheme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.19638","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-28T17:52:24Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"21dbecd3b8894fdf9d899bfb0321163c6fa40ada1b4f60225ed4691821440a47","abstract_canon_sha256":"03998c89fb95e656adcf4ed01da555450a9d7881f5370768dfa0f93628520687"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:01:56.250532Z","signature_b64":"kzjkZqCDyxXjarp8u+MeHUw23GzM0mMSNZW/WnztluwtxNB9hULKM6ZO0RVzUE34kIuyEDPkgqyqnwO734xvDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cacf7c2bcb449e5fc6bca3dd8202e48b178dd2a2ac4b4792b8cc916250398559","last_reissued_at":"2026-07-05T08:01:56.250059Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:01:56.250059Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Siamese Vision Transformers are Scalable Audio-visual Learners","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Gedas Bertasius, Yan-Bo Lin","submitted_at":"2024-03-28T17:52:24Z","abstract_excerpt":"Traditional audio-visual methods rely on independent audio and visual backbones, which is costly and not scalable. In this work, we investigate using an audio-visual siamese network (AVSiam) for efficient and scalable audio-visual pretraining. Our framework uses a single shared vision transformer backbone to process audio and visual inputs, improving its parameter efficiency, reducing the GPU memory footprint, and allowing us to scale our method to larger datasets and model sizes. We pretrain our model using a contrastive audio-visual matching objective with a multi-ratio random masking scheme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.19638","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.19638/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.19638","created_at":"2026-07-05T08:01:56.250120+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.19638v1","created_at":"2026-07-05T08:01:56.250120+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.19638","created_at":"2026-07-05T08:01:56.250120+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZLHXYK6LISPF","created_at":"2026-07-05T08:01:56.250120+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZLHXYK6LISPF7RV4","created_at":"2026-07-05T08:01:56.250120+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZLHXYK6L","created_at":"2026-07-05T08:01:56.250120+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25225","citing_title":"MJEPA: A Simple and Scalable Joint-Embedding Predictive Architecture for Audio-Visual Learning","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZLHXYK6LISPF7RV4UPOYEAXERM","json":"https://pith.science/pith/ZLHXYK6LISPF7RV4UPOYEAXERM.json","graph_json":"https://pith.science/api/pith-number/ZLHXYK6LISPF7RV4UPOYEAXERM/graph.json","events_json":"https://pith.science/api/pith-number/ZLHXYK6LISPF7RV4UPOYEAXERM/events.json","paper":"https://pith.science/paper/ZLHXYK6L"},"agent_actions":{"view_html":"https://pith.science/pith/ZLHXYK6LISPF7RV4UPOYEAXERM","download_json":"https://pith.science/pith/ZLHXYK6LISPF7RV4UPOYEAXERM.json","view_paper":"https://pith.science/paper/ZLHXYK6L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.19638&json=true","fetch_graph":"https://pith.science/api/pith-number/ZLHXYK6LISPF7RV4UPOYEAXERM/graph.json","fetch_events":"https://pith.science/api/pith-number/ZLHXYK6LISPF7RV4UPOYEAXERM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZLHXYK6LISPF7RV4UPOYEAXERM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZLHXYK6LISPF7RV4UPOYEAXERM/action/storage_attestation","attest_author":"https://pith.science/pith/ZLHXYK6LISPF7RV4UPOYEAXERM/action/author_attestation","sign_citation":"https://pith.science/pith/ZLHXYK6LISPF7RV4UPOYEAXERM/action/citation_signature","submit_replication":"https://pith.science/pith/ZLHXYK6LISPF7RV4UPOYEAXERM/action/replication_record"}},"created_at":"2026-07-05T08:01:56.250120+00:00","updated_at":"2026-07-05T08:01:56.250120+00:00"}