{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PLSCFWQ7V6C32CVUGA6ZHKO4FC","short_pith_number":"pith:PLSCFWQ7","schema_version":"1.0","canonical_sha256":"7ae422da1faf85bd0ab4303d93a9dc28a73720c30eaf7191be31714e7843998d","source":{"kind":"arxiv","id":"2412.05515","version":1},"attestation_state":"computed","paper":{"title":"Video2Reward: Generating Reward Function from Videos for Legged Robot Behavior Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Changxin Huang, Dingjie Zhou, Fuchun Sun, Hui Li, Jianqiang Li, Junlin Liu, Qiwei Liang, Runhao Zeng, Xiping Hu","submitted_at":"2024-12-07T03:10:27Z","abstract_excerpt":"Learning behavior in legged robots presents a significant challenge due to its inherent instability and complex constraints. Recent research has proposed the use of a large language model (LLM) to generate reward functions in reinforcement learning, thereby replacing the need for manually designed rewards by experts. However, this approach, which relies on textual descriptions to define learning objectives, fails to achieve controllable and precise behavior learning with clear directionality. In this paper, we introduce a new video2reward method, which directly generates reward functions from "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.05515","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-12-07T03:10:27Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"77d568fb23cd8a5ab17c1f371a5958eae322613a66a420e57c799650b8d9a7f3","abstract_canon_sha256":"b96b86e97e10d91c6e0eb2186c2b8237e854a23779ee929765a2a30c7c295e52"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:32.755843Z","signature_b64":"F76esGTAuqoG2x5UAvwxEUb5sjmKGB+ythH4BIqJg/YM7P67utvefbY55/e/y+3SghJ4H2s76dmHCTr+UcwWDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7ae422da1faf85bd0ab4303d93a9dc28a73720c30eaf7191be31714e7843998d","last_reissued_at":"2026-07-05T11:28:32.755328Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:32.755328Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video2Reward: Generating Reward Function from Videos for Legged Robot Behavior Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Changxin Huang, Dingjie Zhou, Fuchun Sun, Hui Li, Jianqiang Li, Junlin Liu, Qiwei Liang, Runhao Zeng, Xiping Hu","submitted_at":"2024-12-07T03:10:27Z","abstract_excerpt":"Learning behavior in legged robots presents a significant challenge due to its inherent instability and complex constraints. Recent research has proposed the use of a large language model (LLM) to generate reward functions in reinforcement learning, thereby replacing the need for manually designed rewards by experts. However, this approach, which relies on textual descriptions to define learning objectives, fails to achieve controllable and precise behavior learning with clear directionality. In this paper, we introduce a new video2reward method, which directly generates reward functions from "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.05515","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.05515/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.05515","created_at":"2026-07-05T11:28:32.755393+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.05515v1","created_at":"2026-07-05T11:28:32.755393+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.05515","created_at":"2026-07-05T11:28:32.755393+00:00"},{"alias_kind":"pith_short_12","alias_value":"PLSCFWQ7V6C3","created_at":"2026-07-05T11:28:32.755393+00:00"},{"alias_kind":"pith_short_16","alias_value":"PLSCFWQ7V6C32CVU","created_at":"2026-07-05T11:28:32.755393+00:00"},{"alias_kind":"pith_short_8","alias_value":"PLSCFWQ7","created_at":"2026-07-05T11:28:32.755393+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22123","citing_title":"Beyond Pixels: Learning Invariant Rewards for Real-World Robotics From a Few Demonstrations","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PLSCFWQ7V6C32CVUGA6ZHKO4FC","json":"https://pith.science/pith/PLSCFWQ7V6C32CVUGA6ZHKO4FC.json","graph_json":"https://pith.science/api/pith-number/PLSCFWQ7V6C32CVUGA6ZHKO4FC/graph.json","events_json":"https://pith.science/api/pith-number/PLSCFWQ7V6C32CVUGA6ZHKO4FC/events.json","paper":"https://pith.science/paper/PLSCFWQ7"},"agent_actions":{"view_html":"https://pith.science/pith/PLSCFWQ7V6C32CVUGA6ZHKO4FC","download_json":"https://pith.science/pith/PLSCFWQ7V6C32CVUGA6ZHKO4FC.json","view_paper":"https://pith.science/paper/PLSCFWQ7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.05515&json=true","fetch_graph":"https://pith.science/api/pith-number/PLSCFWQ7V6C32CVUGA6ZHKO4FC/graph.json","fetch_events":"https://pith.science/api/pith-number/PLSCFWQ7V6C32CVUGA6ZHKO4FC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PLSCFWQ7V6C32CVUGA6ZHKO4FC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PLSCFWQ7V6C32CVUGA6ZHKO4FC/action/storage_attestation","attest_author":"https://pith.science/pith/PLSCFWQ7V6C32CVUGA6ZHKO4FC/action/author_attestation","sign_citation":"https://pith.science/pith/PLSCFWQ7V6C32CVUGA6ZHKO4FC/action/citation_signature","submit_replication":"https://pith.science/pith/PLSCFWQ7V6C32CVUGA6ZHKO4FC/action/replication_record"}},"created_at":"2026-07-05T11:28:32.755393+00:00","updated_at":"2026-07-05T11:28:32.755393+00:00"}