{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P4WS4P2GHCTYJLIOSLVGGYWUDL","short_pith_number":"pith:P4WS4P2G","schema_version":"1.0","canonical_sha256":"7f2d2e3f4638a784ad0e92ea6362d41ad267a56d5b7ebb522da43e278e3fecc8","source":{"kind":"arxiv","id":"2402.13659","version":2},"attestation_state":"computed","paper":{"title":"Privacy-Preserving Instructions for Aligning Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Da Yu, Peter Kairouz, Sewoong Oh, Zheng Xu","submitted_at":"2024-02-21T09:45:08Z","abstract_excerpt":"Service providers of large language model (LLM) applications collect user instructions in the wild and use them in further aligning LLMs with users' intentions. These instructions, which potentially contain sensitive information, are annotated by human workers in the process. This poses a new privacy risk not addressed by the typical private optimization. To this end, we propose using synthetic instructions to replace real instructions in data annotation and model fine-tuning. Formal differential privacy is guaranteed by generating those synthetic instructions using privately fine-tuned genera"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.13659","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-02-21T09:45:08Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"b18d4b73738ae3cebe31680f84ff231fd5ee79cb068cd6cfdebf0a07f6acd013","abstract_canon_sha256":"fc218b8cd7efa51cc272795c454cfc2f7471366eaa88ff04dcea4bc7e5a31963"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:57.680836Z","signature_b64":"VXqJW5ZzfuO5M6/FHkcKRZKp7wopp+u4Qh/Oqsim7qLw9P+WL+SQvRH19RZ/AWEfZc09M9KSkLDHrL/C8cgoCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f2d2e3f4638a784ad0e92ea6362d41ad267a56d5b7ebb522da43e278e3fecc8","last_reissued_at":"2026-07-05T08:38:57.680431Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:57.680431Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Privacy-Preserving Instructions for Aligning Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Da Yu, Peter Kairouz, Sewoong Oh, Zheng Xu","submitted_at":"2024-02-21T09:45:08Z","abstract_excerpt":"Service providers of large language model (LLM) applications collect user instructions in the wild and use them in further aligning LLMs with users' intentions. These instructions, which potentially contain sensitive information, are annotated by human workers in the process. This poses a new privacy risk not addressed by the typical private optimization. To this end, we propose using synthetic instructions to replace real instructions in data annotation and model fine-tuning. Formal differential privacy is guaranteed by generating those synthetic instructions using privately fine-tuned genera"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.13659","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.13659/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.13659","created_at":"2026-07-05T08:38:57.680486+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.13659v2","created_at":"2026-07-05T08:38:57.680486+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.13659","created_at":"2026-07-05T08:38:57.680486+00:00"},{"alias_kind":"pith_short_12","alias_value":"P4WS4P2GHCTY","created_at":"2026-07-05T08:38:57.680486+00:00"},{"alias_kind":"pith_short_16","alias_value":"P4WS4P2GHCTYJLIO","created_at":"2026-07-05T08:38:57.680486+00:00"},{"alias_kind":"pith_short_8","alias_value":"P4WS4P2G","created_at":"2026-07-05T08:38:57.680486+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09145","citing_title":"PrivCode++: Latent-Conditioned Differentially Private Code Generation for Comprehensive Guarantees","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02974","citing_title":"InvisibleInk: High-Utility and Low-Cost Text Generation with Differential Privacy","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2506.02153","citing_title":"Small Language Models are the Future of Agentic AI","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12160","citing_title":"PubSwap: Public-Data Off-Policy Coordination for Federated RLVR","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P4WS4P2GHCTYJLIOSLVGGYWUDL","json":"https://pith.science/pith/P4WS4P2GHCTYJLIOSLVGGYWUDL.json","graph_json":"https://pith.science/api/pith-number/P4WS4P2GHCTYJLIOSLVGGYWUDL/graph.json","events_json":"https://pith.science/api/pith-number/P4WS4P2GHCTYJLIOSLVGGYWUDL/events.json","paper":"https://pith.science/paper/P4WS4P2G"},"agent_actions":{"view_html":"https://pith.science/pith/P4WS4P2GHCTYJLIOSLVGGYWUDL","download_json":"https://pith.science/pith/P4WS4P2GHCTYJLIOSLVGGYWUDL.json","view_paper":"https://pith.science/paper/P4WS4P2G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.13659&json=true","fetch_graph":"https://pith.science/api/pith-number/P4WS4P2GHCTYJLIOSLVGGYWUDL/graph.json","fetch_events":"https://pith.science/api/pith-number/P4WS4P2GHCTYJLIOSLVGGYWUDL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P4WS4P2GHCTYJLIOSLVGGYWUDL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P4WS4P2GHCTYJLIOSLVGGYWUDL/action/storage_attestation","attest_author":"https://pith.science/pith/P4WS4P2GHCTYJLIOSLVGGYWUDL/action/author_attestation","sign_citation":"https://pith.science/pith/P4WS4P2GHCTYJLIOSLVGGYWUDL/action/citation_signature","submit_replication":"https://pith.science/pith/P4WS4P2GHCTYJLIOSLVGGYWUDL/action/replication_record"}},"created_at":"2026-07-05T08:38:57.680486+00:00","updated_at":"2026-07-05T08:38:57.680486+00:00"}