{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JAJZYIRNGZCAZXZ4AFTD7NCKWS","short_pith_number":"pith:JAJZYIRN","schema_version":"1.0","canonical_sha256":"48139c222d36440cdf3c01663fb44ab4a1e46ed10132762f62f1d838b6ddeee2","source":{"kind":"arxiv","id":"2405.20340","version":1},"attestation_state":"computed","paper":{"title":"MotionLLM: Understanding Human Behaviors from Human Motions and Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ailing Zeng, Benyou Wang, Hao Zhang, Lei Zhang, Ling-Hao Chen, Ruimao Zhang, Shunlin Lu","submitted_at":"2024-05-30T17:59:50Z","abstract_excerpt":"This study delves into the realm of multi-modality (i.e., video and motion modalities) human behavior understanding by leveraging the powerful capabilities of Large Language Models (LLMs). Diverging from recent LLMs designed for video-only or motion-only understanding, we argue that understanding human behavior necessitates joint modeling from both videos and motion sequences (e.g., SMPL sequences) to capture nuanced body part dynamics and semantics effectively. In light of this, we present MotionLLM, a straightforward yet effective framework for human motion understanding, captioning, and rea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.20340","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-30T17:59:50Z","cross_cats_sorted":[],"title_canon_sha256":"168c85f102d8ad0569ad7c0d57d3e83a7b2b51987fbb14c32453ad41d374ea95","abstract_canon_sha256":"25ceaf81e41ab9ba4b442260a7c02f198186dacb05eaf3e1173f6b985b221be3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:24.597748Z","signature_b64":"mUoFKgvFA7WwrfcK+sAsmRqYgsHW/5uYfvN5mnh6pW5wFNRi+eItCES/2ZmnTEQ3ZI0sPO/HhsXS3Lw7Tm0nAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"48139c222d36440cdf3c01663fb44ab4a1e46ed10132762f62f1d838b6ddeee2","last_reissued_at":"2026-07-05T08:25:24.597254Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:24.597254Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MotionLLM: Understanding Human Behaviors from Human Motions and Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ailing Zeng, Benyou Wang, Hao Zhang, Lei Zhang, Ling-Hao Chen, Ruimao Zhang, Shunlin Lu","submitted_at":"2024-05-30T17:59:50Z","abstract_excerpt":"This study delves into the realm of multi-modality (i.e., video and motion modalities) human behavior understanding by leveraging the powerful capabilities of Large Language Models (LLMs). Diverging from recent LLMs designed for video-only or motion-only understanding, we argue that understanding human behavior necessitates joint modeling from both videos and motion sequences (e.g., SMPL sequences) to capture nuanced body part dynamics and semantics effectively. In light of this, we present MotionLLM, a straightforward yet effective framework for human motion understanding, captioning, and rea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20340","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.20340/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.20340","created_at":"2026-07-05T08:25:24.597327+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.20340v1","created_at":"2026-07-05T08:25:24.597327+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20340","created_at":"2026-07-05T08:25:24.597327+00:00"},{"alias_kind":"pith_short_12","alias_value":"JAJZYIRNGZCA","created_at":"2026-07-05T08:25:24.597327+00:00"},{"alias_kind":"pith_short_16","alias_value":"JAJZYIRNGZCAZXZ4","created_at":"2026-07-05T08:25:24.597327+00:00"},{"alias_kind":"pith_short_8","alias_value":"JAJZYIRN","created_at":"2026-07-05T08:25:24.597327+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20888","citing_title":"Fine-grained Human Motion Understanding with Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04773","citing_title":"NextMotionQA: Benchmarking and Judging Human Motion Understanding with Vision-Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18373","citing_title":"MASS: Motion-Aware Spatial-Temporal Grounding for Physics Reasoning and Comprehension in Vision-Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14070","citing_title":"WirelessSenseLLM: Zero-Shot Human Activity Understanding by Bridging Wireless Signals and Human Language","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23264","citing_title":"MotionHiFlow: Text-to-motion via hierarchical flow matching","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21926","citing_title":"Seeing Without Eyes: 4D Human-Scene Understanding from Wearable IMUs","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JAJZYIRNGZCAZXZ4AFTD7NCKWS","json":"https://pith.science/pith/JAJZYIRNGZCAZXZ4AFTD7NCKWS.json","graph_json":"https://pith.science/api/pith-number/JAJZYIRNGZCAZXZ4AFTD7NCKWS/graph.json","events_json":"https://pith.science/api/pith-number/JAJZYIRNGZCAZXZ4AFTD7NCKWS/events.json","paper":"https://pith.science/paper/JAJZYIRN"},"agent_actions":{"view_html":"https://pith.science/pith/JAJZYIRNGZCAZXZ4AFTD7NCKWS","download_json":"https://pith.science/pith/JAJZYIRNGZCAZXZ4AFTD7NCKWS.json","view_paper":"https://pith.science/paper/JAJZYIRN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.20340&json=true","fetch_graph":"https://pith.science/api/pith-number/JAJZYIRNGZCAZXZ4AFTD7NCKWS/graph.json","fetch_events":"https://pith.science/api/pith-number/JAJZYIRNGZCAZXZ4AFTD7NCKWS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JAJZYIRNGZCAZXZ4AFTD7NCKWS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JAJZYIRNGZCAZXZ4AFTD7NCKWS/action/storage_attestation","attest_author":"https://pith.science/pith/JAJZYIRNGZCAZXZ4AFTD7NCKWS/action/author_attestation","sign_citation":"https://pith.science/pith/JAJZYIRNGZCAZXZ4AFTD7NCKWS/action/citation_signature","submit_replication":"https://pith.science/pith/JAJZYIRNGZCAZXZ4AFTD7NCKWS/action/replication_record"}},"created_at":"2026-07-05T08:25:24.597327+00:00","updated_at":"2026-07-05T08:25:24.597327+00:00"}