{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YGLAVIO2M2ALDDZKNZ7TL56OEY","short_pith_number":"pith:YGLAVIO2","schema_version":"1.0","canonical_sha256":"c1960aa1da6680b18f2a6e7f35f7ce263a1db96ae04df6399b6ad1df448b6e06","source":{"kind":"arxiv","id":"2502.17322","version":1},"attestation_state":"computed","paper":{"title":"TDMPBC: Self-Imitative Reinforcement Learning for Humanoid Robot Control","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Diyuan Shi, Donglin Wang, Hongyin Zhang, Runze Suo, Shangke Lyu, Ting Wang, Xiao He, Zifeng Zhuang","submitted_at":"2025-02-24T16:55:27Z","abstract_excerpt":"Complex high-dimensional spaces with high Degree-of-Freedom and complicated action spaces, such as humanoid robots equipped with dexterous hands, pose significant challenges for reinforcement learning (RL) algorithms, which need to wisely balance exploration and exploitation under limited sample budgets. In general, feasible regions for accomplishing tasks within complex high-dimensional spaces are exceedingly narrow. For instance, in the context of humanoid robot motion control, the vast majority of space corresponds to falling, while only a minuscule fraction corresponds to standing upright,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.17322","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-02-24T16:55:27Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b9ecc447676bc0789a9f9b9f6a4a0aac323d56ae131031df9fbb5961c1434e0b","abstract_canon_sha256":"cc44741a7b267a6b1787d1ee7e8a5642dcf0cb5335944d304238bf8cce3c5d0d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:08.739388Z","signature_b64":"6b3HQSm0iZ0Z5hPTdw9+5UYHjSQMUfoVy+pLxIn+/RhVbuLUp1HCadS9TwGk28XRymQCyJINQmdZNKVzTbe/CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c1960aa1da6680b18f2a6e7f35f7ce263a1db96ae04df6399b6ad1df448b6e06","last_reissued_at":"2026-07-05T10:19:08.738925Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:08.738925Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TDMPBC: Self-Imitative Reinforcement Learning for Humanoid Robot Control","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Diyuan Shi, Donglin Wang, Hongyin Zhang, Runze Suo, Shangke Lyu, Ting Wang, Xiao He, Zifeng Zhuang","submitted_at":"2025-02-24T16:55:27Z","abstract_excerpt":"Complex high-dimensional spaces with high Degree-of-Freedom and complicated action spaces, such as humanoid robots equipped with dexterous hands, pose significant challenges for reinforcement learning (RL) algorithms, which need to wisely balance exploration and exploitation under limited sample budgets. In general, feasible regions for accomplishing tasks within complex high-dimensional spaces are exceedingly narrow. For instance, in the context of humanoid robot motion control, the vast majority of space corresponds to falling, while only a minuscule fraction corresponds to standing upright,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.17322","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.17322/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.17322","created_at":"2026-07-05T10:19:08.738973+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.17322v1","created_at":"2026-07-05T10:19:08.738973+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.17322","created_at":"2026-07-05T10:19:08.738973+00:00"},{"alias_kind":"pith_short_12","alias_value":"YGLAVIO2M2AL","created_at":"2026-07-05T10:19:08.738973+00:00"},{"alias_kind":"pith_short_16","alias_value":"YGLAVIO2M2ALDDZK","created_at":"2026-07-05T10:19:08.738973+00:00"},{"alias_kind":"pith_short_8","alias_value":"YGLAVIO2","created_at":"2026-07-05T10:19:08.738973+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.12612","citing_title":"FastDSAC: Unlocking the Potential of Maximum Entropy RL in High-Dimensional Humanoid Control","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YGLAVIO2M2ALDDZKNZ7TL56OEY","json":"https://pith.science/pith/YGLAVIO2M2ALDDZKNZ7TL56OEY.json","graph_json":"https://pith.science/api/pith-number/YGLAVIO2M2ALDDZKNZ7TL56OEY/graph.json","events_json":"https://pith.science/api/pith-number/YGLAVIO2M2ALDDZKNZ7TL56OEY/events.json","paper":"https://pith.science/paper/YGLAVIO2"},"agent_actions":{"view_html":"https://pith.science/pith/YGLAVIO2M2ALDDZKNZ7TL56OEY","download_json":"https://pith.science/pith/YGLAVIO2M2ALDDZKNZ7TL56OEY.json","view_paper":"https://pith.science/paper/YGLAVIO2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.17322&json=true","fetch_graph":"https://pith.science/api/pith-number/YGLAVIO2M2ALDDZKNZ7TL56OEY/graph.json","fetch_events":"https://pith.science/api/pith-number/YGLAVIO2M2ALDDZKNZ7TL56OEY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YGLAVIO2M2ALDDZKNZ7TL56OEY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YGLAVIO2M2ALDDZKNZ7TL56OEY/action/storage_attestation","attest_author":"https://pith.science/pith/YGLAVIO2M2ALDDZKNZ7TL56OEY/action/author_attestation","sign_citation":"https://pith.science/pith/YGLAVIO2M2ALDDZKNZ7TL56OEY/action/citation_signature","submit_replication":"https://pith.science/pith/YGLAVIO2M2ALDDZKNZ7TL56OEY/action/replication_record"}},"created_at":"2026-07-05T10:19:08.738973+00:00","updated_at":"2026-07-05T10:19:08.738973+00:00"}