{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RTNKDT7G2SEQM6VHYOYPMJNUZ5","short_pith_number":"pith:RTNKDT7G","schema_version":"1.0","canonical_sha256":"8cdaa1cfe6d489067aa7c3b0f625b4cf4290745fa6f5820213c3252c8430e8c9","source":{"kind":"arxiv","id":"2405.03689","version":2},"attestation_state":"computed","paper":{"title":"Pose Priors from Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dan Klein, Evonne Ng, Lea M\\\"uller, Sanjay Subramanian, Shiry Ginosar, Trevor Darrell","submitted_at":"2024-05-06T17:59:36Z","abstract_excerpt":"Language is often used to describe physical interaction, yet most 3D human pose estimation methods overlook this rich source of information. We bridge this gap by leveraging large multimodal models (LMMs) as priors for reconstructing contact poses, offering a scalable alternative to traditional methods that rely on human annotations or motion capture data. Our approach extracts contact-relevant descriptors from an LMM and translates them into tractable losses to constrain 3D human pose optimization. Despite its simplicity, our method produces compelling reconstructions for both two-person inte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.03689","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-06T17:59:36Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d28a564365b066cf59ae366809cf95e85bc32a917f0c3b97a30d8e1c9a1b19ba","abstract_canon_sha256":"62e4b62921007ad3efcfe6d6920d6ec71a057f26b503a05c1e23adcbcedf9ce4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:14.476804Z","signature_b64":"1gjcyMxdZZWzWP2bgRn2C+mfPoXfg1OpgD1Gw66jCYTdFa8BOx2D7HuW0Ns0carlRwuG14L7J4t8RblwNZ+JBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8cdaa1cfe6d489067aa7c3b0f625b4cf4290745fa6f5820213c3252c8430e8c9","last_reissued_at":"2026-07-05T11:03:14.476308Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:14.476308Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pose Priors from Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dan Klein, Evonne Ng, Lea M\\\"uller, Sanjay Subramanian, Shiry Ginosar, Trevor Darrell","submitted_at":"2024-05-06T17:59:36Z","abstract_excerpt":"Language is often used to describe physical interaction, yet most 3D human pose estimation methods overlook this rich source of information. We bridge this gap by leveraging large multimodal models (LMMs) as priors for reconstructing contact poses, offering a scalable alternative to traditional methods that rely on human annotations or motion capture data. Our approach extracts contact-relevant descriptors from an LMM and translates them into tractable losses to constrain 3D human pose optimization. Despite its simplicity, our method produces compelling reconstructions for both two-person inte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.03689","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.03689/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.03689","created_at":"2026-07-05T11:03:14.476381+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.03689v2","created_at":"2026-07-05T11:03:14.476381+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.03689","created_at":"2026-07-05T11:03:14.476381+00:00"},{"alias_kind":"pith_short_12","alias_value":"RTNKDT7G2SEQ","created_at":"2026-07-05T11:03:14.476381+00:00"},{"alias_kind":"pith_short_16","alias_value":"RTNKDT7G2SEQM6VH","created_at":"2026-07-05T11:03:14.476381+00:00"},{"alias_kind":"pith_short_8","alias_value":"RTNKDT7G","created_at":"2026-07-05T11:03:14.476381+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.13581","citing_title":"SocialMirror: Reconstructing 3D Human Interaction Behaviors from Monocular Videos with Semantic and Geometric Guidance","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RTNKDT7G2SEQM6VHYOYPMJNUZ5","json":"https://pith.science/pith/RTNKDT7G2SEQM6VHYOYPMJNUZ5.json","graph_json":"https://pith.science/api/pith-number/RTNKDT7G2SEQM6VHYOYPMJNUZ5/graph.json","events_json":"https://pith.science/api/pith-number/RTNKDT7G2SEQM6VHYOYPMJNUZ5/events.json","paper":"https://pith.science/paper/RTNKDT7G"},"agent_actions":{"view_html":"https://pith.science/pith/RTNKDT7G2SEQM6VHYOYPMJNUZ5","download_json":"https://pith.science/pith/RTNKDT7G2SEQM6VHYOYPMJNUZ5.json","view_paper":"https://pith.science/paper/RTNKDT7G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.03689&json=true","fetch_graph":"https://pith.science/api/pith-number/RTNKDT7G2SEQM6VHYOYPMJNUZ5/graph.json","fetch_events":"https://pith.science/api/pith-number/RTNKDT7G2SEQM6VHYOYPMJNUZ5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RTNKDT7G2SEQM6VHYOYPMJNUZ5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RTNKDT7G2SEQM6VHYOYPMJNUZ5/action/storage_attestation","attest_author":"https://pith.science/pith/RTNKDT7G2SEQM6VHYOYPMJNUZ5/action/author_attestation","sign_citation":"https://pith.science/pith/RTNKDT7G2SEQM6VHYOYPMJNUZ5/action/citation_signature","submit_replication":"https://pith.science/pith/RTNKDT7G2SEQM6VHYOYPMJNUZ5/action/replication_record"}},"created_at":"2026-07-05T11:03:14.476381+00:00","updated_at":"2026-07-05T11:03:14.476381+00:00"}