{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O7DUKAO2TWCFAZ3HPAWU3ZROUE","short_pith_number":"pith:O7DUKAO2","schema_version":"1.0","canonical_sha256":"77c74501da9d84506767782d4de62ea12dfa2341ff52643b5d11aaa259e209da","source":{"kind":"arxiv","id":"2407.13095","version":1},"attestation_state":"computed","paper":{"title":"Audio-visual Generalized Zero-shot Learning the Easy Way","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Pedro Morgado, Shentong Mo","submitted_at":"2024-07-18T01:57:16Z","abstract_excerpt":"Audio-visual generalized zero-shot learning is a rapidly advancing domain that seeks to understand the intricate relations between audio and visual cues within videos. The overarching goal is to leverage insights from seen classes to identify instances from previously unseen ones. Prior approaches primarily utilized synchronized auto-encoders to reconstruct audio-visual attributes, which were informed by cross-attention transformers and projected text embeddings. However, these methods fell short of effectively capturing the intricate relationship between cross-modal features and class-label e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.13095","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-18T01:57:16Z","cross_cats_sorted":["cs.LG","cs.MM","cs.SD","eess.AS"],"title_canon_sha256":"2451cfa84622860359ae4e4ebf9ae5098e8f526f62ce3a3fe3fd14c2d6ccc917","abstract_canon_sha256":"85bf26391e4ae582a985d7c542da72c01562eb72714e32afb08216d3f2396925"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:33.441449Z","signature_b64":"dALNIh7P7yYW/lMg8cOOUpgnfBwnXr4Awy1foYrdQzdC9z+pIdEmLa2oOgnJma4aEWE6ij31UnMamOQp5s4ECQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77c74501da9d84506767782d4de62ea12dfa2341ff52643b5d11aaa259e209da","last_reissued_at":"2026-07-05T08:45:33.440985Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:33.440985Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Audio-visual Generalized Zero-shot Learning the Easy Way","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Pedro Morgado, Shentong Mo","submitted_at":"2024-07-18T01:57:16Z","abstract_excerpt":"Audio-visual generalized zero-shot learning is a rapidly advancing domain that seeks to understand the intricate relations between audio and visual cues within videos. The overarching goal is to leverage insights from seen classes to identify instances from previously unseen ones. Prior approaches primarily utilized synchronized auto-encoders to reconstruct audio-visual attributes, which were informed by cross-attention transformers and projected text embeddings. However, these methods fell short of effectively capturing the intricate relationship between cross-modal features and class-label e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.13095","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.13095/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.13095","created_at":"2026-07-05T08:45:33.441041+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.13095v1","created_at":"2026-07-05T08:45:33.441041+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.13095","created_at":"2026-07-05T08:45:33.441041+00:00"},{"alias_kind":"pith_short_12","alias_value":"O7DUKAO2TWCF","created_at":"2026-07-05T08:45:33.441041+00:00"},{"alias_kind":"pith_short_16","alias_value":"O7DUKAO2TWCFAZ3H","created_at":"2026-07-05T08:45:33.441041+00:00"},{"alias_kind":"pith_short_8","alias_value":"O7DUKAO2","created_at":"2026-07-05T08:45:33.441041+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O7DUKAO2TWCFAZ3HPAWU3ZROUE","json":"https://pith.science/pith/O7DUKAO2TWCFAZ3HPAWU3ZROUE.json","graph_json":"https://pith.science/api/pith-number/O7DUKAO2TWCFAZ3HPAWU3ZROUE/graph.json","events_json":"https://pith.science/api/pith-number/O7DUKAO2TWCFAZ3HPAWU3ZROUE/events.json","paper":"https://pith.science/paper/O7DUKAO2"},"agent_actions":{"view_html":"https://pith.science/pith/O7DUKAO2TWCFAZ3HPAWU3ZROUE","download_json":"https://pith.science/pith/O7DUKAO2TWCFAZ3HPAWU3ZROUE.json","view_paper":"https://pith.science/paper/O7DUKAO2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.13095&json=true","fetch_graph":"https://pith.science/api/pith-number/O7DUKAO2TWCFAZ3HPAWU3ZROUE/graph.json","fetch_events":"https://pith.science/api/pith-number/O7DUKAO2TWCFAZ3HPAWU3ZROUE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O7DUKAO2TWCFAZ3HPAWU3ZROUE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O7DUKAO2TWCFAZ3HPAWU3ZROUE/action/storage_attestation","attest_author":"https://pith.science/pith/O7DUKAO2TWCFAZ3HPAWU3ZROUE/action/author_attestation","sign_citation":"https://pith.science/pith/O7DUKAO2TWCFAZ3HPAWU3ZROUE/action/citation_signature","submit_replication":"https://pith.science/pith/O7DUKAO2TWCFAZ3HPAWU3ZROUE/action/replication_record"}},"created_at":"2026-07-05T08:45:33.441041+00:00","updated_at":"2026-07-05T08:45:33.441041+00:00"}