{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:Q7NR3W66PQAEL4GMDWD52NSDN5","short_pith_number":"pith:Q7NR3W66","schema_version":"1.0","canonical_sha256":"87db1ddbde7c0045f0cc1d87dd36436f4e289a9e746fabee47d96050914b71e6","source":{"kind":"arxiv","id":"2310.14092","version":1},"attestation_state":"computed","paper":{"title":"Learning Reward for Physical Skills using Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Yiqing Xu, Yuwei Zeng","submitted_at":"2023-10-21T19:10:06Z","abstract_excerpt":"Learning reward functions for physical skills are challenging due to the vast spectrum of skills, the high-dimensionality of state and action space, and nuanced sensory feedback. The complexity of these tasks makes acquiring expert demonstration data both costly and time-consuming. Large Language Models (LLMs) contain valuable task-related knowledge that can aid in learning these reward functions. However, the direct application of LLMs for proposing reward functions has its limitations such as numerical instability and inability to incorporate the environment feedback. We aim to extract task "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.14092","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-10-21T19:10:06Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f315bec09071fecb8093b3b18e1ca84f4651656085d1ec4615181c42799ac339","abstract_canon_sha256":"e3190cd29a8233b5a0b14b507121e15daab75881e527dbc6c7ce650bb4c6e299"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:03:39.528155Z","signature_b64":"uRy9BAbRObh/AZURBvXSu/ujzQAKEcpPYT378oMpQlx2Ub6XRdAgwncECKErq4iLfHDYPqSJkjD36gRcx4fwCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"87db1ddbde7c0045f0cc1d87dd36436f4e289a9e746fabee47d96050914b71e6","last_reissued_at":"2026-07-05T07:03:39.527674Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:03:39.527674Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Reward for Physical Skills using Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Yiqing Xu, Yuwei Zeng","submitted_at":"2023-10-21T19:10:06Z","abstract_excerpt":"Learning reward functions for physical skills are challenging due to the vast spectrum of skills, the high-dimensionality of state and action space, and nuanced sensory feedback. The complexity of these tasks makes acquiring expert demonstration data both costly and time-consuming. Large Language Models (LLMs) contain valuable task-related knowledge that can aid in learning these reward functions. However, the direct application of LLMs for proposing reward functions has its limitations such as numerical instability and inability to incorporate the environment feedback. We aim to extract task "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.14092","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.14092/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.14092","created_at":"2026-07-05T07:03:39.527729+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.14092v1","created_at":"2026-07-05T07:03:39.527729+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.14092","created_at":"2026-07-05T07:03:39.527729+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q7NR3W66PQAE","created_at":"2026-07-05T07:03:39.527729+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q7NR3W66PQAEL4GM","created_at":"2026-07-05T07:03:39.527729+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q7NR3W66","created_at":"2026-07-05T07:03:39.527729+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2307.05973","citing_title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","ref_index":86,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q7NR3W66PQAEL4GMDWD52NSDN5","json":"https://pith.science/pith/Q7NR3W66PQAEL4GMDWD52NSDN5.json","graph_json":"https://pith.science/api/pith-number/Q7NR3W66PQAEL4GMDWD52NSDN5/graph.json","events_json":"https://pith.science/api/pith-number/Q7NR3W66PQAEL4GMDWD52NSDN5/events.json","paper":"https://pith.science/paper/Q7NR3W66"},"agent_actions":{"view_html":"https://pith.science/pith/Q7NR3W66PQAEL4GMDWD52NSDN5","download_json":"https://pith.science/pith/Q7NR3W66PQAEL4GMDWD52NSDN5.json","view_paper":"https://pith.science/paper/Q7NR3W66","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.14092&json=true","fetch_graph":"https://pith.science/api/pith-number/Q7NR3W66PQAEL4GMDWD52NSDN5/graph.json","fetch_events":"https://pith.science/api/pith-number/Q7NR3W66PQAEL4GMDWD52NSDN5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q7NR3W66PQAEL4GMDWD52NSDN5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q7NR3W66PQAEL4GMDWD52NSDN5/action/storage_attestation","attest_author":"https://pith.science/pith/Q7NR3W66PQAEL4GMDWD52NSDN5/action/author_attestation","sign_citation":"https://pith.science/pith/Q7NR3W66PQAEL4GMDWD52NSDN5/action/citation_signature","submit_replication":"https://pith.science/pith/Q7NR3W66PQAEL4GMDWD52NSDN5/action/replication_record"}},"created_at":"2026-07-05T07:03:39.527729+00:00","updated_at":"2026-07-05T07:03:39.527729+00:00"}