{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JE7TCFSFSOAFQCZIY3CCIYCSWG","short_pith_number":"pith:JE7TCFSF","schema_version":"1.0","canonical_sha256":"493f3116459380580b28c6c4246052b19f9298e7c92370a4c5919d67256f1829","source":{"kind":"arxiv","id":"2405.07162","version":3},"attestation_state":"computed","paper":{"title":"Learning Reward for Robot Skills Using Large Language Models via Self-Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Lin Shao, Yao Mu, Yuwei Zeng","submitted_at":"2024-05-12T04:57:43Z","abstract_excerpt":"Learning reward functions remains the bottleneck to equip a robot with a broad repertoire of skills. Large Language Models (LLM) contain valuable task-related knowledge that can potentially aid in the learning of reward functions. However, the proposed reward function can be imprecise, thus ineffective which requires to be further grounded with environment information. We proposed a method to learn rewards more efficiently in the absence of humans. Our approach consists of two components: We first use the LLM to propose features and parameterization of the reward, then update the parameters th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.07162","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-05-12T04:57:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4ebaee99303352ec2f654684f7e3441452196e81038350897ba89d425f9b15f3","abstract_canon_sha256":"8bb9444f40d9c83c7d78e64626da09918192b1ae6a49573f8ffff173bbedd1ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:19:45.064324Z","signature_b64":"IEOmcqZ6xnhQOue3+7C6jVUG6FINg6vyRcooCL3fezntn7DPKYNFZgBnoyRRdVzumEK+G6Gq3OGz/Sknc0nSCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"493f3116459380580b28c6c4246052b19f9298e7c92370a4c5919d67256f1829","last_reissued_at":"2026-07-05T08:19:45.063753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:19:45.063753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Reward for Robot Skills Using Large Language Models via Self-Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Lin Shao, Yao Mu, Yuwei Zeng","submitted_at":"2024-05-12T04:57:43Z","abstract_excerpt":"Learning reward functions remains the bottleneck to equip a robot with a broad repertoire of skills. Large Language Models (LLM) contain valuable task-related knowledge that can potentially aid in the learning of reward functions. However, the proposed reward function can be imprecise, thus ineffective which requires to be further grounded with environment information. We proposed a method to learn rewards more efficiently in the absence of humans. Our approach consists of two components: We first use the LLM to propose features and parameterization of the reward, then update the parameters th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.07162","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.07162/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.07162","created_at":"2026-07-05T08:19:45.063818+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.07162v3","created_at":"2026-07-05T08:19:45.063818+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.07162","created_at":"2026-07-05T08:19:45.063818+00:00"},{"alias_kind":"pith_short_12","alias_value":"JE7TCFSFSOAF","created_at":"2026-07-05T08:19:45.063818+00:00"},{"alias_kind":"pith_short_16","alias_value":"JE7TCFSFSOAFQCZI","created_at":"2026-07-05T08:19:45.063818+00:00"},{"alias_kind":"pith_short_8","alias_value":"JE7TCFSF","created_at":"2026-07-05T08:19:45.063818+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25398","citing_title":"MAPL: Multi-Objective Preference Learning for Robot Locomotion","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2409.02428","citing_title":"Language Models as Efficient Reward Function Searchers for Custom-Environment Multi-Objective Reinforcement","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19259","citing_title":"ERFSL: An Efficient Reward Function Searcher via Language Models for Custom-Environment Multi-Objective Optimization (Student Abstract)","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JE7TCFSFSOAFQCZIY3CCIYCSWG","json":"https://pith.science/pith/JE7TCFSFSOAFQCZIY3CCIYCSWG.json","graph_json":"https://pith.science/api/pith-number/JE7TCFSFSOAFQCZIY3CCIYCSWG/graph.json","events_json":"https://pith.science/api/pith-number/JE7TCFSFSOAFQCZIY3CCIYCSWG/events.json","paper":"https://pith.science/paper/JE7TCFSF"},"agent_actions":{"view_html":"https://pith.science/pith/JE7TCFSFSOAFQCZIY3CCIYCSWG","download_json":"https://pith.science/pith/JE7TCFSFSOAFQCZIY3CCIYCSWG.json","view_paper":"https://pith.science/paper/JE7TCFSF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.07162&json=true","fetch_graph":"https://pith.science/api/pith-number/JE7TCFSFSOAFQCZIY3CCIYCSWG/graph.json","fetch_events":"https://pith.science/api/pith-number/JE7TCFSFSOAFQCZIY3CCIYCSWG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JE7TCFSFSOAFQCZIY3CCIYCSWG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JE7TCFSFSOAFQCZIY3CCIYCSWG/action/storage_attestation","attest_author":"https://pith.science/pith/JE7TCFSFSOAFQCZIY3CCIYCSWG/action/author_attestation","sign_citation":"https://pith.science/pith/JE7TCFSFSOAFQCZIY3CCIYCSWG/action/citation_signature","submit_replication":"https://pith.science/pith/JE7TCFSFSOAFQCZIY3CCIYCSWG/action/replication_record"}},"created_at":"2026-07-05T08:19:45.063818+00:00","updated_at":"2026-07-05T08:19:45.063818+00:00"}