{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PMBYILRCS2SBW3QTCG4NJA475O","short_pith_number":"pith:PMBYILRC","schema_version":"1.0","canonical_sha256":"7b03842e2296a41b6e1311b8d4839feb8a7cf502a8ef01ad7257971c722dcda8","source":{"kind":"arxiv","id":"2505.01425","version":1},"attestation_state":"computed","paper":{"title":"GENMO: A GENeralist Model for Human MOtion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.RO"],"primary_cat":"cs.GR","authors_text":"Davis Rempe, Haotian Zhang, Jan Kautz, Jiefeng Li, Jinkun Cao, Umar Iqbal, Ye Yuan","submitted_at":"2025-05-02T17:59:55Z","abstract_excerpt":"Human motion modeling traditionally separates motion generation and estimation into distinct tasks with specialized models. Motion generation models focus on creating diverse, realistic motions from inputs like text, audio, or keyframes, while motion estimation models aim to reconstruct accurate motion trajectories from observations like videos. Despite sharing underlying representations of temporal dynamics and kinematics, this separation limits knowledge transfer between tasks and requires maintaining separate models. We present GENMO, a unified Generalist Model for Human Motion that bridges"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.01425","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.GR","submitted_at":"2025-05-02T17:59:55Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG","cs.RO"],"title_canon_sha256":"fbc54871612b05ed3576c6cdc58baa729d5de5720c8d49dd888740ecf48250ef","abstract_canon_sha256":"d85089e17a8f42a66abc86e2f44bc15169cf456814690062578d626b4c2f03f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:57:50.776082Z","signature_b64":"2gR/5yUE1wvK2jHIzW68mvkWSBpyPgMY0PoEOBy7fT2MBhBDos8Kar03h4M4WFOYPZ7jvZcNOqsoiwYhOXGqAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b03842e2296a41b6e1311b8d4839feb8a7cf502a8ef01ad7257971c722dcda8","last_reissued_at":"2026-07-05T10:57:50.775593Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:57:50.775593Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GENMO: A GENeralist Model for Human MOtion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.RO"],"primary_cat":"cs.GR","authors_text":"Davis Rempe, Haotian Zhang, Jan Kautz, Jiefeng Li, Jinkun Cao, Umar Iqbal, Ye Yuan","submitted_at":"2025-05-02T17:59:55Z","abstract_excerpt":"Human motion modeling traditionally separates motion generation and estimation into distinct tasks with specialized models. Motion generation models focus on creating diverse, realistic motions from inputs like text, audio, or keyframes, while motion estimation models aim to reconstruct accurate motion trajectories from observations like videos. Despite sharing underlying representations of temporal dynamics and kinematics, this separation limits knowledge transfer between tasks and requires maintaining separate models. We present GENMO, a unified Generalist Model for Human Motion that bridges"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.01425","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.01425/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.01425","created_at":"2026-07-05T10:57:50.775654+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.01425v1","created_at":"2026-07-05T10:57:50.775654+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.01425","created_at":"2026-07-05T10:57:50.775654+00:00"},{"alias_kind":"pith_short_12","alias_value":"PMBYILRCS2SB","created_at":"2026-07-05T10:57:50.775654+00:00"},{"alias_kind":"pith_short_16","alias_value":"PMBYILRCS2SBW3QT","created_at":"2026-07-05T10:57:50.775654+00:00"},{"alias_kind":"pith_short_8","alias_value":"PMBYILRC","created_at":"2026-07-05T10:57:50.775654+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26215","citing_title":"TaskNPoint: How to Teach Your Humanoid to Hit a Backhand in Minutes","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22726","citing_title":"Text Dictates, Music Decorates: Energy-based Attention for Editable Dance Motion Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18243","citing_title":"MOCHI: Motion Enhancement of Collaborative Human-object Interactions","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13364","citing_title":"VideoMDM: Towards 3D Human Motion Generation From 2D Supervision","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14854","citing_title":"FactorizedHMR: A Hybrid Framework for Video Human Mesh Recovery","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2512.11988","citing_title":"CARI4D: Category Agnostic 4D Reconstruction of Human-Object Interaction","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27711","citing_title":"ExoActor: Exocentric Video Generation as Generalizable Interactive Humanoid Control","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05390","citing_title":"LAMP: Localization Aware Multi-camera People Tracking in Metric 3D World","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01234","citing_title":"TT4D: A Pipeline and Dataset for Table Tennis 4D Reconstruction From Monocular Videos","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21926","citing_title":"Seeing Without Eyes: 4D Human-Scene Understanding from Wearable IMUs","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10466","citing_title":"ExpertEdit: Learning Skill-Aware Motion Editing from Expert Videos","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PMBYILRCS2SBW3QTCG4NJA475O","json":"https://pith.science/pith/PMBYILRCS2SBW3QTCG4NJA475O.json","graph_json":"https://pith.science/api/pith-number/PMBYILRCS2SBW3QTCG4NJA475O/graph.json","events_json":"https://pith.science/api/pith-number/PMBYILRCS2SBW3QTCG4NJA475O/events.json","paper":"https://pith.science/paper/PMBYILRC"},"agent_actions":{"view_html":"https://pith.science/pith/PMBYILRCS2SBW3QTCG4NJA475O","download_json":"https://pith.science/pith/PMBYILRCS2SBW3QTCG4NJA475O.json","view_paper":"https://pith.science/paper/PMBYILRC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.01425&json=true","fetch_graph":"https://pith.science/api/pith-number/PMBYILRCS2SBW3QTCG4NJA475O/graph.json","fetch_events":"https://pith.science/api/pith-number/PMBYILRCS2SBW3QTCG4NJA475O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PMBYILRCS2SBW3QTCG4NJA475O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PMBYILRCS2SBW3QTCG4NJA475O/action/storage_attestation","attest_author":"https://pith.science/pith/PMBYILRCS2SBW3QTCG4NJA475O/action/author_attestation","sign_citation":"https://pith.science/pith/PMBYILRCS2SBW3QTCG4NJA475O/action/citation_signature","submit_replication":"https://pith.science/pith/PMBYILRCS2SBW3QTCG4NJA475O/action/replication_record"}},"created_at":"2026-07-05T10:57:50.775654+00:00","updated_at":"2026-07-05T10:57:50.775654+00:00"}