{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NXJABLLF2432Q2WVHMI7L2T2AQ","short_pith_number":"pith:NXJABLLF","schema_version":"1.0","canonical_sha256":"6dd200ad65d737a86ad53b11f5ea7a043f4803cf2042d0c5d5b37162e45b3d0e","source":{"kind":"arxiv","id":"2407.13460","version":1},"attestation_state":"computed","paper":{"title":"SA-DVAE: Improving Zero-Shot Skeleton-Based Action Recognition by Disentangled Variational Autoencoders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Chih-Yuan Yang, Jane Yung-Jen Hsu, Sheng-Wei Li, Wei-Jie Chen, Yi-Hsin Yu, Zi-Xiang Wei","submitted_at":"2024-07-18T12:35:46Z","abstract_excerpt":"Existing zero-shot skeleton-based action recognition methods utilize projection networks to learn a shared latent space of skeleton features and semantic embeddings. The inherent imbalance in action recognition datasets, characterized by variable skeleton sequences yet constant class labels, presents significant challenges for alignment. To address the imbalance, we propose SA-DVAE -- Semantic Alignment via Disentangled Variational Autoencoders, a method that first adopts feature disentanglement to separate skeleton features into two independent parts -- one is semantic-related and another is "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.13460","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-18T12:35:46Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"f16fb263d977a45633ef43a9c9b4e992af499f46e9c7d6fffa9dd8d0f6a30e0a","abstract_canon_sha256":"ec4a6f2d54678126e3867cafde0f79ef7b5c7da31a0d63d097018ed0ad1f9108"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:39.250597Z","signature_b64":"26sC1mlS7iiVm1IZmIodRLD+W13JQ+x+AiK4uCsVvxjkYPPmMporuVRGHvh8m5g5H0BmEsK2pag0mbAe9+FLCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6dd200ad65d737a86ad53b11f5ea7a043f4803cf2042d0c5d5b37162e45b3d0e","last_reissued_at":"2026-07-05T08:45:39.250172Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:39.250172Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SA-DVAE: Improving Zero-Shot Skeleton-Based Action Recognition by Disentangled Variational Autoencoders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Chih-Yuan Yang, Jane Yung-Jen Hsu, Sheng-Wei Li, Wei-Jie Chen, Yi-Hsin Yu, Zi-Xiang Wei","submitted_at":"2024-07-18T12:35:46Z","abstract_excerpt":"Existing zero-shot skeleton-based action recognition methods utilize projection networks to learn a shared latent space of skeleton features and semantic embeddings. The inherent imbalance in action recognition datasets, characterized by variable skeleton sequences yet constant class labels, presents significant challenges for alignment. To address the imbalance, we propose SA-DVAE -- Semantic Alignment via Disentangled Variational Autoencoders, a method that first adopts feature disentanglement to separate skeleton features into two independent parts -- one is semantic-related and another is "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.13460","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.13460/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.13460","created_at":"2026-07-05T08:45:39.250233+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.13460v1","created_at":"2026-07-05T08:45:39.250233+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.13460","created_at":"2026-07-05T08:45:39.250233+00:00"},{"alias_kind":"pith_short_12","alias_value":"NXJABLLF2432","created_at":"2026-07-05T08:45:39.250233+00:00"},{"alias_kind":"pith_short_16","alias_value":"NXJABLLF2432Q2WV","created_at":"2026-07-05T08:45:39.250233+00:00"},{"alias_kind":"pith_short_8","alias_value":"NXJABLLF","created_at":"2026-07-05T08:45:39.250233+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.18003","citing_title":"Universal Skeleton Understanding via Differentiable Rendering and MLLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18003","citing_title":"Universal Skeleton Understanding via Differentiable Rendering and MLLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18003","citing_title":"Universal Skeleton Understanding via Differentiable Rendering and MLLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09063","citing_title":"Frequency-Enhanced Diffusion Models: Curriculum-Guided Semantic Alignment for Zero-Shot Skeleton Action Recognition","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NXJABLLF2432Q2WVHMI7L2T2AQ","json":"https://pith.science/pith/NXJABLLF2432Q2WVHMI7L2T2AQ.json","graph_json":"https://pith.science/api/pith-number/NXJABLLF2432Q2WVHMI7L2T2AQ/graph.json","events_json":"https://pith.science/api/pith-number/NXJABLLF2432Q2WVHMI7L2T2AQ/events.json","paper":"https://pith.science/paper/NXJABLLF"},"agent_actions":{"view_html":"https://pith.science/pith/NXJABLLF2432Q2WVHMI7L2T2AQ","download_json":"https://pith.science/pith/NXJABLLF2432Q2WVHMI7L2T2AQ.json","view_paper":"https://pith.science/paper/NXJABLLF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.13460&json=true","fetch_graph":"https://pith.science/api/pith-number/NXJABLLF2432Q2WVHMI7L2T2AQ/graph.json","fetch_events":"https://pith.science/api/pith-number/NXJABLLF2432Q2WVHMI7L2T2AQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NXJABLLF2432Q2WVHMI7L2T2AQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NXJABLLF2432Q2WVHMI7L2T2AQ/action/storage_attestation","attest_author":"https://pith.science/pith/NXJABLLF2432Q2WVHMI7L2T2AQ/action/author_attestation","sign_citation":"https://pith.science/pith/NXJABLLF2432Q2WVHMI7L2T2AQ/action/citation_signature","submit_replication":"https://pith.science/pith/NXJABLLF2432Q2WVHMI7L2T2AQ/action/replication_record"}},"created_at":"2026-07-05T08:45:39.250233+00:00","updated_at":"2026-07-05T08:45:39.250233+00:00"}