{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YYJWP2VBXKN42JEOLKZZAYLNUN","short_pith_number":"pith:YYJWP2VB","schema_version":"1.0","canonical_sha256":"c61367eaa1ba9bcd248e5ab390616da37d1ee7e702cfb3ec9cd9fc009580de1d","source":{"kind":"arxiv","id":"2304.04602","version":2},"attestation_state":"computed","paper":{"title":"Learning a Universal Human Prior for Dexterous Manipulation from Human Preference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC","cs.LG"],"primary_cat":"cs.RO","authors_text":"Allen Z. Ren, Chi Jin, Hao Dong, Qianxu Wang, Shixiang Shane Gu, Yuanpei Chen, Zihan Ding","submitted_at":"2023-04-10T14:17:33Z","abstract_excerpt":"Generating human-like behavior on robots is a great challenge especially in dexterous manipulation tasks with robotic hands. Scripting policies from scratch is intractable due to the high-dimensional control space, and training policies with reinforcement learning (RL) and manual reward engineering can also be hard and lead to unnatural motions. Leveraging the recent progress on RL from Human Feedback, we propose a framework that learns a universal human prior using direct human preference feedback over videos, for efficiently tuning the RL policies on 20 dual-hand robot manipulation tasks in "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.04602","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2023-04-10T14:17:33Z","cross_cats_sorted":["cs.HC","cs.LG"],"title_canon_sha256":"698799f085e8bc67def5893aa5307b997b2ae3ad8dfa454813b167204c3c5928","abstract_canon_sha256":"e9e1315ec9d40cd857139f2b4b1282b42b9880bc8d0ec6c04684276e4172866f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:50:13.802659Z","signature_b64":"2RUxPWc5d2iuGxZk+ghHvUwpU5MaHHpCRG1+dlGDNSAs7Fe5Vn7ft9+MWgI7PRDbN519LudtexcNSBQgVNJ8Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c61367eaa1ba9bcd248e5ab390616da37d1ee7e702cfb3ec9cd9fc009580de1d","last_reissued_at":"2026-07-05T06:50:13.802127Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:50:13.802127Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning a Universal Human Prior for Dexterous Manipulation from Human Preference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC","cs.LG"],"primary_cat":"cs.RO","authors_text":"Allen Z. Ren, Chi Jin, Hao Dong, Qianxu Wang, Shixiang Shane Gu, Yuanpei Chen, Zihan Ding","submitted_at":"2023-04-10T14:17:33Z","abstract_excerpt":"Generating human-like behavior on robots is a great challenge especially in dexterous manipulation tasks with robotic hands. Scripting policies from scratch is intractable due to the high-dimensional control space, and training policies with reinforcement learning (RL) and manual reward engineering can also be hard and lead to unnatural motions. Leveraging the recent progress on RL from Human Feedback, we propose a framework that learns a universal human prior using direct human preference feedback over videos, for efficiently tuning the RL policies on 20 dual-hand robot manipulation tasks in "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.04602","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.04602/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.04602","created_at":"2026-07-05T06:50:13.802184+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.04602v2","created_at":"2026-07-05T06:50:13.802184+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.04602","created_at":"2026-07-05T06:50:13.802184+00:00"},{"alias_kind":"pith_short_12","alias_value":"YYJWP2VBXKN4","created_at":"2026-07-05T06:50:13.802184+00:00"},{"alias_kind":"pith_short_16","alias_value":"YYJWP2VBXKN42JEO","created_at":"2026-07-05T06:50:13.802184+00:00"},{"alias_kind":"pith_short_8","alias_value":"YYJWP2VB","created_at":"2026-07-05T06:50:13.802184+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.14317","citing_title":"ClutterDexGrasp: A Sim-to-Real System for General Dexterous Grasping in Cluttered Scenes","ref_index":37,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YYJWP2VBXKN42JEOLKZZAYLNUN","json":"https://pith.science/pith/YYJWP2VBXKN42JEOLKZZAYLNUN.json","graph_json":"https://pith.science/api/pith-number/YYJWP2VBXKN42JEOLKZZAYLNUN/graph.json","events_json":"https://pith.science/api/pith-number/YYJWP2VBXKN42JEOLKZZAYLNUN/events.json","paper":"https://pith.science/paper/YYJWP2VB"},"agent_actions":{"view_html":"https://pith.science/pith/YYJWP2VBXKN42JEOLKZZAYLNUN","download_json":"https://pith.science/pith/YYJWP2VBXKN42JEOLKZZAYLNUN.json","view_paper":"https://pith.science/paper/YYJWP2VB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.04602&json=true","fetch_graph":"https://pith.science/api/pith-number/YYJWP2VBXKN42JEOLKZZAYLNUN/graph.json","fetch_events":"https://pith.science/api/pith-number/YYJWP2VBXKN42JEOLKZZAYLNUN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YYJWP2VBXKN42JEOLKZZAYLNUN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YYJWP2VBXKN42JEOLKZZAYLNUN/action/storage_attestation","attest_author":"https://pith.science/pith/YYJWP2VBXKN42JEOLKZZAYLNUN/action/author_attestation","sign_citation":"https://pith.science/pith/YYJWP2VBXKN42JEOLKZZAYLNUN/action/citation_signature","submit_replication":"https://pith.science/pith/YYJWP2VBXKN42JEOLKZZAYLNUN/action/replication_record"}},"created_at":"2026-07-05T06:50:13.802184+00:00","updated_at":"2026-07-05T06:50:13.802184+00:00"}