{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5T3KR7YEEKMSTJ2KDVXWXJIKQA","short_pith_number":"pith:5T3KR7YE","schema_version":"1.0","canonical_sha256":"ecf6a8ff04229929a74a1d6f6ba50a80160bccc19cf928a01eed14528289889a","source":{"kind":"arxiv","id":"2504.08654","version":1},"attestation_state":"computed","paper":{"title":"The Invisible EgoHand: 3D Hand Forecasting through EgoBody Pose Estimation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dima Damen, Hideo Saito, Masashi Hatano, Zhifan Zhu","submitted_at":"2025-04-11T15:58:31Z","abstract_excerpt":"Forecasting hand motion and pose from an egocentric perspective is essential for understanding human intention. However, existing methods focus solely on predicting positions without considering articulation, and only when the hands are visible in the field of view. This limitation overlooks the fact that approximate hand positions can still be inferred even when they are outside the camera's view. In this paper, we propose a method to forecast the 3D trajectories and poses of both hands from an egocentric video, both in and out of the field of view. We propose a diffusion-based transformer ar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.08654","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-11T15:58:31Z","cross_cats_sorted":[],"title_canon_sha256":"8302eb005e17dda75eb17074a5a54516d7f4c546ebd95794af5f7897405ee478","abstract_canon_sha256":"b51de4c4c6a3d02f44e4936e0b8412a566b3f5ad485f27af750dd929c541bbe4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:52.343066Z","signature_b64":"HYYg3i95DJ+u5vpE8Ug3CWu2OxNeF1wR5TRQ4aR6+C2DZRfHyBan1EQxJyHZzvVibLVtbcHlHoZ4hRLDinYhCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ecf6a8ff04229929a74a1d6f6ba50a80160bccc19cf928a01eed14528289889a","last_reissued_at":"2026-07-05T10:47:52.342599Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:52.342599Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Invisible EgoHand: 3D Hand Forecasting through EgoBody Pose Estimation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dima Damen, Hideo Saito, Masashi Hatano, Zhifan Zhu","submitted_at":"2025-04-11T15:58:31Z","abstract_excerpt":"Forecasting hand motion and pose from an egocentric perspective is essential for understanding human intention. However, existing methods focus solely on predicting positions without considering articulation, and only when the hands are visible in the field of view. This limitation overlooks the fact that approximate hand positions can still be inferred even when they are outside the camera's view. In this paper, we propose a method to forecast the 3D trajectories and poses of both hands from an egocentric video, both in and out of the field of view. We propose a diffusion-based transformer ar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.08654","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.08654/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.08654","created_at":"2026-07-05T10:47:52.342657+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.08654v1","created_at":"2026-07-05T10:47:52.342657+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.08654","created_at":"2026-07-05T10:47:52.342657+00:00"},{"alias_kind":"pith_short_12","alias_value":"5T3KR7YEEKMS","created_at":"2026-07-05T10:47:52.342657+00:00"},{"alias_kind":"pith_short_16","alias_value":"5T3KR7YEEKMSTJ2K","created_at":"2026-07-05T10:47:52.342657+00:00"},{"alias_kind":"pith_short_8","alias_value":"5T3KR7YE","created_at":"2026-07-05T10:47:52.342657+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07001","citing_title":"Ego-Human Motion Prediction with 3D-Aware LLM","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05938","citing_title":"Prior-First, Condition-Second: Scalable and Controllable Hand Motion Completion","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2511.18127","citing_title":"SFHand: Learning Embodied Manipulation by Streaming Egocentric 3D Hand Forecasting","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2511.12878","citing_title":"Uni-Hand: Universal Hand Motion Forecasting in Egocentric Views","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07642","citing_title":"EggHand: A Multimodal Foundation Model for Egocentric Hand Pose Forecasting","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5T3KR7YEEKMSTJ2KDVXWXJIKQA","json":"https://pith.science/pith/5T3KR7YEEKMSTJ2KDVXWXJIKQA.json","graph_json":"https://pith.science/api/pith-number/5T3KR7YEEKMSTJ2KDVXWXJIKQA/graph.json","events_json":"https://pith.science/api/pith-number/5T3KR7YEEKMSTJ2KDVXWXJIKQA/events.json","paper":"https://pith.science/paper/5T3KR7YE"},"agent_actions":{"view_html":"https://pith.science/pith/5T3KR7YEEKMSTJ2KDVXWXJIKQA","download_json":"https://pith.science/pith/5T3KR7YEEKMSTJ2KDVXWXJIKQA.json","view_paper":"https://pith.science/paper/5T3KR7YE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.08654&json=true","fetch_graph":"https://pith.science/api/pith-number/5T3KR7YEEKMSTJ2KDVXWXJIKQA/graph.json","fetch_events":"https://pith.science/api/pith-number/5T3KR7YEEKMSTJ2KDVXWXJIKQA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5T3KR7YEEKMSTJ2KDVXWXJIKQA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5T3KR7YEEKMSTJ2KDVXWXJIKQA/action/storage_attestation","attest_author":"https://pith.science/pith/5T3KR7YEEKMSTJ2KDVXWXJIKQA/action/author_attestation","sign_citation":"https://pith.science/pith/5T3KR7YEEKMSTJ2KDVXWXJIKQA/action/citation_signature","submit_replication":"https://pith.science/pith/5T3KR7YEEKMSTJ2KDVXWXJIKQA/action/replication_record"}},"created_at":"2026-07-05T10:47:52.342657+00:00","updated_at":"2026-07-05T10:47:52.342657+00:00"}