{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IZZLG4APLFC3YVCQR3FWHW2OXP","short_pith_number":"pith:IZZLG4AP","schema_version":"1.0","canonical_sha256":"4672b3700f5945bc54508ecb63db4ebbe2b88feaf826ab46a20dfe4b1ec1e308","source":{"kind":"arxiv","id":"2407.12108","version":2},"attestation_state":"computed","paper":{"title":"Private prediction for large-scale synthetic text generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Alex Bie, Alexey Kurakin, Andreas Terzis, Kareem Amin, Natalia Ponomareva, Sergei Vassilvitskii, Umar Syed, Weiwei Kong","submitted_at":"2024-07-16T18:28:40Z","abstract_excerpt":"We present an approach for generating differentially private synthetic text using large language models (LLMs), via private prediction. In the private prediction framework, we only require the output synthetic data to satisfy differential privacy guarantees. This is in contrast to approaches that train a generative model on potentially sensitive user-supplied source data and seek to ensure the model itself is safe to release.\n  We prompt a pretrained LLM with source data, but ensure that next-token predictions are made with differential privacy guarantees. Previous work in this paradigm report"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.12108","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-16T18:28:40Z","cross_cats_sorted":["cs.CL","cs.CR"],"title_canon_sha256":"9f21712384e71b5ae6fca7fbe200bce4d29cd5ae26f7c72dd42d16c963001b58","abstract_canon_sha256":"189660d902b89ad5123036eb2fb3429cba5ac8da124a7024fad80e35481ae3da"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:46.044164Z","signature_b64":"ZvNS8/gpJVeUXEwRxI44idhFDwu/jN2hwYw0+B8kKJ7onjyyNQ7w89+J4gCZIDtxXWHTg6dG0WC7E6lpICEQBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4672b3700f5945bc54508ecb63db4ebbe2b88feaf826ab46a20dfe4b1ec1e308","last_reissued_at":"2026-07-05T09:17:46.043633Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:46.043633Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Private prediction for large-scale synthetic text generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Alex Bie, Alexey Kurakin, Andreas Terzis, Kareem Amin, Natalia Ponomareva, Sergei Vassilvitskii, Umar Syed, Weiwei Kong","submitted_at":"2024-07-16T18:28:40Z","abstract_excerpt":"We present an approach for generating differentially private synthetic text using large language models (LLMs), via private prediction. In the private prediction framework, we only require the output synthetic data to satisfy differential privacy guarantees. This is in contrast to approaches that train a generative model on potentially sensitive user-supplied source data and seek to ensure the model itself is safe to release.\n  We prompt a pretrained LLM with source data, but ensure that next-token predictions are made with differential privacy guarantees. Previous work in this paradigm report"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.12108","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.12108/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.12108","created_at":"2026-07-05T09:17:46.043694+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.12108v2","created_at":"2026-07-05T09:17:46.043694+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.12108","created_at":"2026-07-05T09:17:46.043694+00:00"},{"alias_kind":"pith_short_12","alias_value":"IZZLG4APLFC3","created_at":"2026-07-05T09:17:46.043694+00:00"},{"alias_kind":"pith_short_16","alias_value":"IZZLG4APLFC3YVCQ","created_at":"2026-07-05T09:17:46.043694+00:00"},{"alias_kind":"pith_short_8","alias_value":"IZZLG4AP","created_at":"2026-07-05T09:17:46.043694+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.02974","citing_title":"InvisibleInk: High-Utility and Low-Cost Text Generation with Differential Privacy","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01425","citing_title":"Barriers to Counterfactual Credit Attribution for Autoregressive Models","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IZZLG4APLFC3YVCQR3FWHW2OXP","json":"https://pith.science/pith/IZZLG4APLFC3YVCQR3FWHW2OXP.json","graph_json":"https://pith.science/api/pith-number/IZZLG4APLFC3YVCQR3FWHW2OXP/graph.json","events_json":"https://pith.science/api/pith-number/IZZLG4APLFC3YVCQR3FWHW2OXP/events.json","paper":"https://pith.science/paper/IZZLG4AP"},"agent_actions":{"view_html":"https://pith.science/pith/IZZLG4APLFC3YVCQR3FWHW2OXP","download_json":"https://pith.science/pith/IZZLG4APLFC3YVCQR3FWHW2OXP.json","view_paper":"https://pith.science/paper/IZZLG4AP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.12108&json=true","fetch_graph":"https://pith.science/api/pith-number/IZZLG4APLFC3YVCQR3FWHW2OXP/graph.json","fetch_events":"https://pith.science/api/pith-number/IZZLG4APLFC3YVCQR3FWHW2OXP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IZZLG4APLFC3YVCQR3FWHW2OXP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IZZLG4APLFC3YVCQR3FWHW2OXP/action/storage_attestation","attest_author":"https://pith.science/pith/IZZLG4APLFC3YVCQR3FWHW2OXP/action/author_attestation","sign_citation":"https://pith.science/pith/IZZLG4APLFC3YVCQR3FWHW2OXP/action/citation_signature","submit_replication":"https://pith.science/pith/IZZLG4APLFC3YVCQR3FWHW2OXP/action/replication_record"}},"created_at":"2026-07-05T09:17:46.043694+00:00","updated_at":"2026-07-05T09:17:46.043694+00:00"}