{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3FPR6JNQIQAMIB67PQK6VCNFGW","short_pith_number":"pith:3FPR6JNQ","schema_version":"1.0","canonical_sha256":"d95f1f25b04400c407df7c15ea89a5359db8217e927e8a0842fc433e9a6cfd38","source":{"kind":"arxiv","id":"2306.04933","version":1},"attestation_state":"computed","paper":{"title":"InfoPrompt: Information-Theoretic Soft Prompt Tuning for Natural Language Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Chaochao Lu, Handong Zhao, Junda Wu, Ricardo Henao, Rui Wang, Ruiyi Zhang, Shuai Li, Tong Yu, Zhao Song","submitted_at":"2023-06-08T04:31:48Z","abstract_excerpt":"Soft prompt tuning achieves superior performances across a wide range of few-shot tasks. However, the performances of prompt tuning can be highly sensitive to the initialization of the prompts. We also empirically observe that conventional prompt tuning methods cannot encode and learn sufficient task-relevant information from prompt tokens. In this work, we develop an information-theoretic framework that formulates soft prompt tuning as maximizing mutual information between prompts and other model parameters (or encoded representations). This novel view helps us to develop a more efficient, ac"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.04933","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-06-08T04:31:48Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"1a37d3bf1e7c4d755699611cc3596a2986531a4eda2d15daa030151cf8f57bbf","abstract_canon_sha256":"d7955248fa99e0d93fc83dbb0069362d1dd3d751184dbefe081790bbbbb049c8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:18:43.618436Z","signature_b64":"yC3ioDOEJtj8AVdqIiwkskEb6+aVZzYzkFfmciWnPQB8SNh93HOh690C/Yq+Ype9lNtOONVv+31KsHgoRTy+Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d95f1f25b04400c407df7c15ea89a5359db8217e927e8a0842fc433e9a6cfd38","last_reissued_at":"2026-07-05T06:18:43.618006Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:18:43.618006Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InfoPrompt: Information-Theoretic Soft Prompt Tuning for Natural Language Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Chaochao Lu, Handong Zhao, Junda Wu, Ricardo Henao, Rui Wang, Ruiyi Zhang, Shuai Li, Tong Yu, Zhao Song","submitted_at":"2023-06-08T04:31:48Z","abstract_excerpt":"Soft prompt tuning achieves superior performances across a wide range of few-shot tasks. However, the performances of prompt tuning can be highly sensitive to the initialization of the prompts. We also empirically observe that conventional prompt tuning methods cannot encode and learn sufficient task-relevant information from prompt tokens. In this work, we develop an information-theoretic framework that formulates soft prompt tuning as maximizing mutual information between prompts and other model parameters (or encoded representations). This novel view helps us to develop a more efficient, ac"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.04933","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.04933/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.04933","created_at":"2026-07-05T06:18:43.618065+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.04933v1","created_at":"2026-07-05T06:18:43.618065+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.04933","created_at":"2026-07-05T06:18:43.618065+00:00"},{"alias_kind":"pith_short_12","alias_value":"3FPR6JNQIQAM","created_at":"2026-07-05T06:18:43.618065+00:00"},{"alias_kind":"pith_short_16","alias_value":"3FPR6JNQIQAMIB67","created_at":"2026-07-05T06:18:43.618065+00:00"},{"alias_kind":"pith_short_8","alias_value":"3FPR6JNQ","created_at":"2026-07-05T06:18:43.618065+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20333","citing_title":"SoftSkill: Behavioral Compression for Contextual Adaptation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2306.14048","citing_title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2403.14608","citing_title":"Parameter-Efficient Fine-Tuning for Large Models: A Comprehensive Survey","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3FPR6JNQIQAMIB67PQK6VCNFGW","json":"https://pith.science/pith/3FPR6JNQIQAMIB67PQK6VCNFGW.json","graph_json":"https://pith.science/api/pith-number/3FPR6JNQIQAMIB67PQK6VCNFGW/graph.json","events_json":"https://pith.science/api/pith-number/3FPR6JNQIQAMIB67PQK6VCNFGW/events.json","paper":"https://pith.science/paper/3FPR6JNQ"},"agent_actions":{"view_html":"https://pith.science/pith/3FPR6JNQIQAMIB67PQK6VCNFGW","download_json":"https://pith.science/pith/3FPR6JNQIQAMIB67PQK6VCNFGW.json","view_paper":"https://pith.science/paper/3FPR6JNQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.04933&json=true","fetch_graph":"https://pith.science/api/pith-number/3FPR6JNQIQAMIB67PQK6VCNFGW/graph.json","fetch_events":"https://pith.science/api/pith-number/3FPR6JNQIQAMIB67PQK6VCNFGW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3FPR6JNQIQAMIB67PQK6VCNFGW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3FPR6JNQIQAMIB67PQK6VCNFGW/action/storage_attestation","attest_author":"https://pith.science/pith/3FPR6JNQIQAMIB67PQK6VCNFGW/action/author_attestation","sign_citation":"https://pith.science/pith/3FPR6JNQIQAMIB67PQK6VCNFGW/action/citation_signature","submit_replication":"https://pith.science/pith/3FPR6JNQIQAMIB67PQK6VCNFGW/action/replication_record"}},"created_at":"2026-07-05T06:18:43.618065+00:00","updated_at":"2026-07-05T06:18:43.618065+00:00"}