{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7SRYFUQFL3KDVOB57IXQAK5YIG","short_pith_number":"pith:7SRYFUQF","schema_version":"1.0","canonical_sha256":"fca382d2055ed43ab83dfa2f002bb841bef1b004c43d418645238dbac7436674","source":{"kind":"arxiv","id":"2406.01316","version":2},"attestation_state":"computed","paper":{"title":"Enhancing Inertial Hand based HAR through Joint Representation of Language, Pose and Synthetic IMUs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Kaishun Wu, Lala Shakti Swarup Ray, Paul Lukowicz, Vitor Fortes Rey, Xia Qingxin","submitted_at":"2024-06-03T13:28:42Z","abstract_excerpt":"Due to the scarcity of labeled sensor data in HAR, prior research has turned to video data to synthesize Inertial Measurement Units (IMU) data, capitalizing on its rich activity annotations. However, generating IMU data from videos presents challenges for HAR in real-world settings, attributed to the poor quality of synthetic IMU data and its limited efficacy in subtle, fine-grained motions. In this paper, we propose Multi$^3$Net, our novel multi-modal, multitask, and contrastive-based framework approach to address the issue of limited data. Our pretraining procedure uses videos from online re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.01316","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-03T13:28:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f8f5b8d8a0133ae9130ef3a3620616d0851bf279ea753829d7cb059107aeda62","abstract_canon_sha256":"4700c218d77a5061c6f76353dfbc510103b42b4d53df063154e38b21280b8fc7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:10.345443Z","signature_b64":"6IYG9tGEYKPO2KfpPkIQ4viJKD6lg32tgVRwpRWogrsArhvT+X+kFSSX4bmSkL3KxDJC1TZmKGEVuflO4b1RAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fca382d2055ed43ab83dfa2f002bb841bef1b004c43d418645238dbac7436674","last_reissued_at":"2026-07-05T08:49:10.344998Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:10.344998Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Inertial Hand based HAR through Joint Representation of Language, Pose and Synthetic IMUs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Kaishun Wu, Lala Shakti Swarup Ray, Paul Lukowicz, Vitor Fortes Rey, Xia Qingxin","submitted_at":"2024-06-03T13:28:42Z","abstract_excerpt":"Due to the scarcity of labeled sensor data in HAR, prior research has turned to video data to synthesize Inertial Measurement Units (IMU) data, capitalizing on its rich activity annotations. However, generating IMU data from videos presents challenges for HAR in real-world settings, attributed to the poor quality of synthetic IMU data and its limited efficacy in subtle, fine-grained motions. In this paper, we propose Multi$^3$Net, our novel multi-modal, multitask, and contrastive-based framework approach to address the issue of limited data. Our pretraining procedure uses videos from online re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.01316","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.01316/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.01316","created_at":"2026-07-05T08:49:10.345054+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.01316v2","created_at":"2026-07-05T08:49:10.345054+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.01316","created_at":"2026-07-05T08:49:10.345054+00:00"},{"alias_kind":"pith_short_12","alias_value":"7SRYFUQFL3KD","created_at":"2026-07-05T08:49:10.345054+00:00"},{"alias_kind":"pith_short_16","alias_value":"7SRYFUQFL3KDVOB5","created_at":"2026-07-05T08:49:10.345054+00:00"},{"alias_kind":"pith_short_8","alias_value":"7SRYFUQF","created_at":"2026-07-05T08:49:10.345054+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.06405","citing_title":"SImpHAR: Advancing impedance-based human activity recognition using 3D simulation and text-to-motion models","ref_index":36,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7SRYFUQFL3KDVOB57IXQAK5YIG","json":"https://pith.science/pith/7SRYFUQFL3KDVOB57IXQAK5YIG.json","graph_json":"https://pith.science/api/pith-number/7SRYFUQFL3KDVOB57IXQAK5YIG/graph.json","events_json":"https://pith.science/api/pith-number/7SRYFUQFL3KDVOB57IXQAK5YIG/events.json","paper":"https://pith.science/paper/7SRYFUQF"},"agent_actions":{"view_html":"https://pith.science/pith/7SRYFUQFL3KDVOB57IXQAK5YIG","download_json":"https://pith.science/pith/7SRYFUQFL3KDVOB57IXQAK5YIG.json","view_paper":"https://pith.science/paper/7SRYFUQF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.01316&json=true","fetch_graph":"https://pith.science/api/pith-number/7SRYFUQFL3KDVOB57IXQAK5YIG/graph.json","fetch_events":"https://pith.science/api/pith-number/7SRYFUQFL3KDVOB57IXQAK5YIG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7SRYFUQFL3KDVOB57IXQAK5YIG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7SRYFUQFL3KDVOB57IXQAK5YIG/action/storage_attestation","attest_author":"https://pith.science/pith/7SRYFUQFL3KDVOB57IXQAK5YIG/action/author_attestation","sign_citation":"https://pith.science/pith/7SRYFUQFL3KDVOB57IXQAK5YIG/action/citation_signature","submit_replication":"https://pith.science/pith/7SRYFUQFL3KDVOB57IXQAK5YIG/action/replication_record"}},"created_at":"2026-07-05T08:49:10.345054+00:00","updated_at":"2026-07-05T08:49:10.345054+00:00"}