{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CT6ZAVMFO5XAGNEUVRL6EO7IFY","short_pith_number":"pith:CT6ZAVMF","schema_version":"1.0","canonical_sha256":"14fd905585776e033494ac57e23be82e2b7253f12e937e59fb4da1098e077c2c","source":{"kind":"arxiv","id":"2411.01168","version":1},"attestation_state":"computed","paper":{"title":"Prompt Tuning with Diffusion for Few-Shot Pre-trained Policy Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dacheng Tao, Li Shen, Shengchao Hu, Wanru Zhao, Weixiong Lin, Ya Zhang","submitted_at":"2024-11-02T07:38:02Z","abstract_excerpt":"Offline reinforcement learning (RL) methods harness previous experiences to derive an optimal policy, forming the foundation for pre-trained large-scale models (PLMs). When encountering tasks not seen before, PLMs often utilize several expert trajectories as prompts to expedite their adaptation to new requirements. Though a range of prompt-tuning methods have been proposed to enhance the quality of prompts, these methods often face optimization restrictions due to prompt initialization, which can significantly constrain the exploration domain and potentially lead to suboptimal solutions. To el"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.01168","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-11-02T07:38:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6404d19efa0e4f6dcaf108f7cbede4ee6e6df44212c2dbcf09e866a4d626c500","abstract_canon_sha256":"f299642929938d0fd0ab78960a14598966f1bbf9e1a7e3532ec69e7938fb8738"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:30:25.991187Z","signature_b64":"cNT3ypuSs4e4w9URWzX8kuCpGezQOnHl9GiZ9nJT4/DfFiWsOSVcc1SXK+jdPMJVkBxQ1gamMfv9NvgCO/o1Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"14fd905585776e033494ac57e23be82e2b7253f12e937e59fb4da1098e077c2c","last_reissued_at":"2026-07-05T09:30:25.990670Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:30:25.990670Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prompt Tuning with Diffusion for Few-Shot Pre-trained Policy Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dacheng Tao, Li Shen, Shengchao Hu, Wanru Zhao, Weixiong Lin, Ya Zhang","submitted_at":"2024-11-02T07:38:02Z","abstract_excerpt":"Offline reinforcement learning (RL) methods harness previous experiences to derive an optimal policy, forming the foundation for pre-trained large-scale models (PLMs). When encountering tasks not seen before, PLMs often utilize several expert trajectories as prompts to expedite their adaptation to new requirements. Though a range of prompt-tuning methods have been proposed to enhance the quality of prompts, these methods often face optimization restrictions due to prompt initialization, which can significantly constrain the exploration domain and potentially lead to suboptimal solutions. To el"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.01168","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.01168/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.01168","created_at":"2026-07-05T09:30:25.990723+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.01168v1","created_at":"2026-07-05T09:30:25.990723+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.01168","created_at":"2026-07-05T09:30:25.990723+00:00"},{"alias_kind":"pith_short_12","alias_value":"CT6ZAVMFO5XA","created_at":"2026-07-05T09:30:25.990723+00:00"},{"alias_kind":"pith_short_16","alias_value":"CT6ZAVMFO5XAGNEU","created_at":"2026-07-05T09:30:25.990723+00:00"},{"alias_kind":"pith_short_8","alias_value":"CT6ZAVMF","created_at":"2026-07-05T09:30:25.990723+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.06358","citing_title":"Prompt-Tuning Bandits: Enabling Few-Shot Generalization for Efficient Multi-Task Offline RL","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CT6ZAVMFO5XAGNEUVRL6EO7IFY","json":"https://pith.science/pith/CT6ZAVMFO5XAGNEUVRL6EO7IFY.json","graph_json":"https://pith.science/api/pith-number/CT6ZAVMFO5XAGNEUVRL6EO7IFY/graph.json","events_json":"https://pith.science/api/pith-number/CT6ZAVMFO5XAGNEUVRL6EO7IFY/events.json","paper":"https://pith.science/paper/CT6ZAVMF"},"agent_actions":{"view_html":"https://pith.science/pith/CT6ZAVMFO5XAGNEUVRL6EO7IFY","download_json":"https://pith.science/pith/CT6ZAVMFO5XAGNEUVRL6EO7IFY.json","view_paper":"https://pith.science/paper/CT6ZAVMF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.01168&json=true","fetch_graph":"https://pith.science/api/pith-number/CT6ZAVMFO5XAGNEUVRL6EO7IFY/graph.json","fetch_events":"https://pith.science/api/pith-number/CT6ZAVMFO5XAGNEUVRL6EO7IFY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CT6ZAVMFO5XAGNEUVRL6EO7IFY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CT6ZAVMFO5XAGNEUVRL6EO7IFY/action/storage_attestation","attest_author":"https://pith.science/pith/CT6ZAVMFO5XAGNEUVRL6EO7IFY/action/author_attestation","sign_citation":"https://pith.science/pith/CT6ZAVMFO5XAGNEUVRL6EO7IFY/action/citation_signature","submit_replication":"https://pith.science/pith/CT6ZAVMFO5XAGNEUVRL6EO7IFY/action/replication_record"}},"created_at":"2026-07-05T09:30:25.990723+00:00","updated_at":"2026-07-05T09:30:25.990723+00:00"}