{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T3JXRDPRP6RLOJY3MWCVQCRU3N","short_pith_number":"pith:T3JXRDPR","schema_version":"1.0","canonical_sha256":"9ed3788df17fa2b7271b6585580a34db570880de1db3629f0c1318ed4248d49e","source":{"kind":"arxiv","id":"2405.04549","version":1},"attestation_state":"computed","paper":{"title":"ClothPPO: A Proximal Policy Optimization Enhancing Framework for Robotic Cloth Manipulation with Observation-Aligned Action Spaces","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CV","authors_text":"Libing Yang, Long Chen, Yang Li","submitted_at":"2024-05-05T12:36:18Z","abstract_excerpt":"Vision-based robotic cloth unfolding has made great progress recently. However, prior works predominantly rely on value learning and have not fully explored policy-based techniques. Recently, the success of reinforcement learning on the large language model has shown that the policy gradient algorithm can enhance policy with huge action space. In this paper, we introduce ClothPPO, a framework that employs a policy gradient algorithm based on actor-critic architecture to enhance a pre-trained model with huge 10^6 action spaces aligned with observation in the task of unfolding clothes. To this e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.04549","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-05T12:36:18Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"210091a0ef801589881db0cdb42c8063a9e702d0c6bf7487612dd1844eea496f","abstract_canon_sha256":"d531efda0015c5fc4de459aef8fa115d3741096dd0695fa790ff1b12c0a56c05"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:16:50.415405Z","signature_b64":"yJAU3ajDHKH9/eX2+ueFMfUSSn0rI86Nsds3f0ZJsRusWWOx7UTX8YzcR3DlcBguq5wvdmSSiX41xn/hquhyAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9ed3788df17fa2b7271b6585580a34db570880de1db3629f0c1318ed4248d49e","last_reissued_at":"2026-07-05T08:16:50.414928Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:16:50.414928Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ClothPPO: A Proximal Policy Optimization Enhancing Framework for Robotic Cloth Manipulation with Observation-Aligned Action Spaces","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CV","authors_text":"Libing Yang, Long Chen, Yang Li","submitted_at":"2024-05-05T12:36:18Z","abstract_excerpt":"Vision-based robotic cloth unfolding has made great progress recently. However, prior works predominantly rely on value learning and have not fully explored policy-based techniques. Recently, the success of reinforcement learning on the large language model has shown that the policy gradient algorithm can enhance policy with huge action space. In this paper, we introduce ClothPPO, a framework that employs a policy gradient algorithm based on actor-critic architecture to enhance a pre-trained model with huge 10^6 action spaces aligned with observation in the task of unfolding clothes. To this e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.04549","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.04549/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.04549","created_at":"2026-07-05T08:16:50.414984+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.04549v1","created_at":"2026-07-05T08:16:50.414984+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.04549","created_at":"2026-07-05T08:16:50.414984+00:00"},{"alias_kind":"pith_short_12","alias_value":"T3JXRDPRP6RL","created_at":"2026-07-05T08:16:50.414984+00:00"},{"alias_kind":"pith_short_16","alias_value":"T3JXRDPRP6RLOJY3","created_at":"2026-07-05T08:16:50.414984+00:00"},{"alias_kind":"pith_short_8","alias_value":"T3JXRDPR","created_at":"2026-07-05T08:16:50.414984+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.22769","citing_title":"Learning Efficient Robotic Garment Manipulation with Standardization","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T3JXRDPRP6RLOJY3MWCVQCRU3N","json":"https://pith.science/pith/T3JXRDPRP6RLOJY3MWCVQCRU3N.json","graph_json":"https://pith.science/api/pith-number/T3JXRDPRP6RLOJY3MWCVQCRU3N/graph.json","events_json":"https://pith.science/api/pith-number/T3JXRDPRP6RLOJY3MWCVQCRU3N/events.json","paper":"https://pith.science/paper/T3JXRDPR"},"agent_actions":{"view_html":"https://pith.science/pith/T3JXRDPRP6RLOJY3MWCVQCRU3N","download_json":"https://pith.science/pith/T3JXRDPRP6RLOJY3MWCVQCRU3N.json","view_paper":"https://pith.science/paper/T3JXRDPR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.04549&json=true","fetch_graph":"https://pith.science/api/pith-number/T3JXRDPRP6RLOJY3MWCVQCRU3N/graph.json","fetch_events":"https://pith.science/api/pith-number/T3JXRDPRP6RLOJY3MWCVQCRU3N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T3JXRDPRP6RLOJY3MWCVQCRU3N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T3JXRDPRP6RLOJY3MWCVQCRU3N/action/storage_attestation","attest_author":"https://pith.science/pith/T3JXRDPRP6RLOJY3MWCVQCRU3N/action/author_attestation","sign_citation":"https://pith.science/pith/T3JXRDPRP6RLOJY3MWCVQCRU3N/action/citation_signature","submit_replication":"https://pith.science/pith/T3JXRDPRP6RLOJY3MWCVQCRU3N/action/replication_record"}},"created_at":"2026-07-05T08:16:50.414984+00:00","updated_at":"2026-07-05T08:16:50.414984+00:00"}