{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:AMXGZKRA3SRTBQWVADL3MUREYJ","short_pith_number":"pith:AMXGZKRA","schema_version":"1.0","canonical_sha256":"032e6caa20dca330c2d500d7b65224c2604e76248c131a146ac88969af145f49","source":{"kind":"arxiv","id":"2305.03568","version":3},"attestation_state":"computed","paper":{"title":"A vector quantized masked autoencoder for audiovisual speech emotion recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Renaud S\\'eguier, Samir Sadok, Simon Leglaive","submitted_at":"2023-05-05T14:19:46Z","abstract_excerpt":"An important challenge in emotion recognition is to develop methods that can leverage unlabeled training data. In this paper, we propose the VQ-MAE-AV model, a self-supervised multimodal model that leverages masked autoencoders to learn representations of audiovisual speech without labels. The model includes vector quantized variational autoencoders that compress raw audio and visual speech data into discrete tokens. The audiovisual speech tokens are used to train a multimodal masked autoencoder that consists of an encoder-decoder architecture with attention mechanisms. The model is designed t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.03568","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2023-05-05T14:19:46Z","cross_cats_sorted":["cs.LG","cs.MM","eess.AS"],"title_canon_sha256":"4ce66f715a2f980a187317c712c8e9ea657aff9d3315458c15e91ea279b5c870","abstract_canon_sha256":"0f378c42f6d1cffe96f4191791fb9cfd34aa88eccfae61b357e02bced130f5f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:00:34.671411Z","signature_b64":"KrN7VbJXWnwZv4wwmCvVIX0B52NcXsZ4NxOls6u9oqGAZOaBB5tNyyupIpCsWGVV9LSHpwFJ3Auct6A+to2oAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"032e6caa20dca330c2d500d7b65224c2604e76248c131a146ac88969af145f49","last_reissued_at":"2026-07-05T11:00:34.670875Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:00:34.670875Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A vector quantized masked autoencoder for audiovisual speech emotion recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Renaud S\\'eguier, Samir Sadok, Simon Leglaive","submitted_at":"2023-05-05T14:19:46Z","abstract_excerpt":"An important challenge in emotion recognition is to develop methods that can leverage unlabeled training data. In this paper, we propose the VQ-MAE-AV model, a self-supervised multimodal model that leverages masked autoencoders to learn representations of audiovisual speech without labels. The model includes vector quantized variational autoencoders that compress raw audio and visual speech data into discrete tokens. The audiovisual speech tokens are used to train a multimodal masked autoencoder that consists of an encoder-decoder architecture with attention mechanisms. The model is designed t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.03568","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.03568/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.03568","created_at":"2026-07-05T11:00:34.670948+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.03568v3","created_at":"2026-07-05T11:00:34.670948+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.03568","created_at":"2026-07-05T11:00:34.670948+00:00"},{"alias_kind":"pith_short_12","alias_value":"AMXGZKRA3SRT","created_at":"2026-07-05T11:00:34.670948+00:00"},{"alias_kind":"pith_short_16","alias_value":"AMXGZKRA3SRTBQWV","created_at":"2026-07-05T11:00:34.670948+00:00"},{"alias_kind":"pith_short_8","alias_value":"AMXGZKRA","created_at":"2026-07-05T11:00:34.670948+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07585","citing_title":"Multimodal Group Emotion Recognition In-the-Wild Towards a Privacy-Safe Non-Individual Approach","ref_index":196,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AMXGZKRA3SRTBQWVADL3MUREYJ","json":"https://pith.science/pith/AMXGZKRA3SRTBQWVADL3MUREYJ.json","graph_json":"https://pith.science/api/pith-number/AMXGZKRA3SRTBQWVADL3MUREYJ/graph.json","events_json":"https://pith.science/api/pith-number/AMXGZKRA3SRTBQWVADL3MUREYJ/events.json","paper":"https://pith.science/paper/AMXGZKRA"},"agent_actions":{"view_html":"https://pith.science/pith/AMXGZKRA3SRTBQWVADL3MUREYJ","download_json":"https://pith.science/pith/AMXGZKRA3SRTBQWVADL3MUREYJ.json","view_paper":"https://pith.science/paper/AMXGZKRA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.03568&json=true","fetch_graph":"https://pith.science/api/pith-number/AMXGZKRA3SRTBQWVADL3MUREYJ/graph.json","fetch_events":"https://pith.science/api/pith-number/AMXGZKRA3SRTBQWVADL3MUREYJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AMXGZKRA3SRTBQWVADL3MUREYJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AMXGZKRA3SRTBQWVADL3MUREYJ/action/storage_attestation","attest_author":"https://pith.science/pith/AMXGZKRA3SRTBQWVADL3MUREYJ/action/author_attestation","sign_citation":"https://pith.science/pith/AMXGZKRA3SRTBQWVADL3MUREYJ/action/citation_signature","submit_replication":"https://pith.science/pith/AMXGZKRA3SRTBQWVADL3MUREYJ/action/replication_record"}},"created_at":"2026-07-05T11:00:34.670948+00:00","updated_at":"2026-07-05T11:00:34.670948+00:00"}