{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4FB3AY7TQHRWW6E226C3O5N7FP","short_pith_number":"pith:4FB3AY7T","schema_version":"1.0","canonical_sha256":"e143b063f381e36b789ad785b775bf2bce5c246f6eb775d3b0c290efb651e863","source":{"kind":"arxiv","id":"2507.18883","version":1},"attestation_state":"computed","paper":{"title":"Success in Humanoid Reinforcement Learning under Partial Observation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.AI","authors_text":"Wuhao Wang, Zhiyong Chen","submitted_at":"2025-07-25T01:51:12Z","abstract_excerpt":"Reinforcement learning has been widely applied to robotic control, but effective policy learning under partial observability remains a major challenge, especially in high-dimensional tasks like humanoid locomotion. To date, no prior work has demonstrated stable training of humanoid policies with incomplete state information in the benchmark Gymnasium Humanoid-v4 environment. The objective in this environment is to walk forward as fast as possible without falling, with rewards provided for staying upright and moving forward, and penalties incurred for excessive actions and external contact forc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.18883","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-07-25T01:51:12Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"18249ff7d2246119c40ffb5082d0f8d932fb7319d465a1f3a5dd47e66cd8acc1","abstract_canon_sha256":"ffa2a0dd04e5807297a9dbde76341ccb3d108fb995108c99986cb066a32590e8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:43:15.388904Z","signature_b64":"GvjzY1ziBQvKqMAXVtmQS5iq7SvV0Z9VLpZXs+iCpUkhXKuDUgKRLunPV1H8+IySOjWUBzTieb7kz7al3qqGCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e143b063f381e36b789ad785b775bf2bce5c246f6eb775d3b0c290efb651e863","last_reissued_at":"2026-07-05T11:43:15.388260Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:43:15.388260Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Success in Humanoid Reinforcement Learning under Partial Observation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.AI","authors_text":"Wuhao Wang, Zhiyong Chen","submitted_at":"2025-07-25T01:51:12Z","abstract_excerpt":"Reinforcement learning has been widely applied to robotic control, but effective policy learning under partial observability remains a major challenge, especially in high-dimensional tasks like humanoid locomotion. To date, no prior work has demonstrated stable training of humanoid policies with incomplete state information in the benchmark Gymnasium Humanoid-v4 environment. The objective in this environment is to walk forward as fast as possible without falling, with rewards provided for staying upright and moving forward, and penalties incurred for excessive actions and external contact forc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.18883","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.18883/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.18883","created_at":"2026-07-05T11:43:15.388375+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.18883v1","created_at":"2026-07-05T11:43:15.388375+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.18883","created_at":"2026-07-05T11:43:15.388375+00:00"},{"alias_kind":"pith_short_12","alias_value":"4FB3AY7TQHRW","created_at":"2026-07-05T11:43:15.388375+00:00"},{"alias_kind":"pith_short_16","alias_value":"4FB3AY7TQHRWW6E2","created_at":"2026-07-05T11:43:15.388375+00:00"},{"alias_kind":"pith_short_8","alias_value":"4FB3AY7T","created_at":"2026-07-05T11:43:15.388375+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4FB3AY7TQHRWW6E226C3O5N7FP","json":"https://pith.science/pith/4FB3AY7TQHRWW6E226C3O5N7FP.json","graph_json":"https://pith.science/api/pith-number/4FB3AY7TQHRWW6E226C3O5N7FP/graph.json","events_json":"https://pith.science/api/pith-number/4FB3AY7TQHRWW6E226C3O5N7FP/events.json","paper":"https://pith.science/paper/4FB3AY7T"},"agent_actions":{"view_html":"https://pith.science/pith/4FB3AY7TQHRWW6E226C3O5N7FP","download_json":"https://pith.science/pith/4FB3AY7TQHRWW6E226C3O5N7FP.json","view_paper":"https://pith.science/paper/4FB3AY7T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.18883&json=true","fetch_graph":"https://pith.science/api/pith-number/4FB3AY7TQHRWW6E226C3O5N7FP/graph.json","fetch_events":"https://pith.science/api/pith-number/4FB3AY7TQHRWW6E226C3O5N7FP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4FB3AY7TQHRWW6E226C3O5N7FP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4FB3AY7TQHRWW6E226C3O5N7FP/action/storage_attestation","attest_author":"https://pith.science/pith/4FB3AY7TQHRWW6E226C3O5N7FP/action/author_attestation","sign_citation":"https://pith.science/pith/4FB3AY7TQHRWW6E226C3O5N7FP/action/citation_signature","submit_replication":"https://pith.science/pith/4FB3AY7TQHRWW6E226C3O5N7FP/action/replication_record"}},"created_at":"2026-07-05T11:43:15.388375+00:00","updated_at":"2026-07-05T11:43:15.388375+00:00"}