{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O4GININTXG55VYYAJBJOLCKCQ7","short_pith_number":"pith:O4GININT","schema_version":"1.0","canonical_sha256":"770c86a1b3b9bbdae3004852e5894287f468f810e4c76ddbfcb4f0e787f378a2","source":{"kind":"arxiv","id":"2409.11423","version":2},"attestation_state":"computed","paper":{"title":"Generated Data with Fake Privacy: Hidden Dangers of Fine-tuning Large Language Models on Generated Data","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Atilla Akkus, Junjie Chu, Masoud Poorghaffar Aghdam, Michael Backes, Mingjie Li, Sinem Sav, Yang Zhang","submitted_at":"2024-09-12T10:14:12Z","abstract_excerpt":"Large language models (LLMs) have demonstrated significant success in various domain-specific tasks, with their performance often improving substantially after fine-tuning. However, fine-tuning with real-world data introduces privacy risks. To mitigate these risks, developers increasingly rely on synthetic data generation as an alternative to using real data, as data generated by traditional models is believed to be different from real-world data. However, with the advanced capabilities of LLMs, the distinction between real data and data generated by these models has become nearly indistinguis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.11423","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CR","submitted_at":"2024-09-12T10:14:12Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"bebc92cbff369e62b7cccf95d0556c63e2216cea2d2d4aad34369f7d72cce823","abstract_canon_sha256":"af9dd852139490bab07cd2f98f1042e0d1998fdba15b34db352ddfc9ad9b79a9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:06:43.854650Z","signature_b64":"hWrsLddJQyOIzt49vNu3NMqi40/kBO+NPjl18Dgr8Ft2oJ8uUCW7QquPSIvvXx+7SDJFlFwfd5ySKWQzjOVEBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"770c86a1b3b9bbdae3004852e5894287f468f810e4c76ddbfcb4f0e787f378a2","last_reissued_at":"2026-07-05T10:06:43.854149Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:06:43.854149Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generated Data with Fake Privacy: Hidden Dangers of Fine-tuning Large Language Models on Generated Data","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Atilla Akkus, Junjie Chu, Masoud Poorghaffar Aghdam, Michael Backes, Mingjie Li, Sinem Sav, Yang Zhang","submitted_at":"2024-09-12T10:14:12Z","abstract_excerpt":"Large language models (LLMs) have demonstrated significant success in various domain-specific tasks, with their performance often improving substantially after fine-tuning. However, fine-tuning with real-world data introduces privacy risks. To mitigate these risks, developers increasingly rely on synthetic data generation as an alternative to using real data, as data generated by traditional models is believed to be different from real-world data. However, with the advanced capabilities of LLMs, the distinction between real data and data generated by these models has become nearly indistinguis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.11423","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.11423/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.11423","created_at":"2026-07-05T10:06:43.854212+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.11423v2","created_at":"2026-07-05T10:06:43.854212+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.11423","created_at":"2026-07-05T10:06:43.854212+00:00"},{"alias_kind":"pith_short_12","alias_value":"O4GININTXG55","created_at":"2026-07-05T10:06:43.854212+00:00"},{"alias_kind":"pith_short_16","alias_value":"O4GININTXG55VYYA","created_at":"2026-07-05T10:06:43.854212+00:00"},{"alias_kind":"pith_short_8","alias_value":"O4GININT","created_at":"2026-07-05T10:06:43.854212+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.14205","citing_title":"Privacy Leakage in Federated Learning in Radiology Reports: A Comparative Evaluation of Tokenizer-Driven Privacy Risks","ref_index":37,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O4GININTXG55VYYAJBJOLCKCQ7","json":"https://pith.science/pith/O4GININTXG55VYYAJBJOLCKCQ7.json","graph_json":"https://pith.science/api/pith-number/O4GININTXG55VYYAJBJOLCKCQ7/graph.json","events_json":"https://pith.science/api/pith-number/O4GININTXG55VYYAJBJOLCKCQ7/events.json","paper":"https://pith.science/paper/O4GININT"},"agent_actions":{"view_html":"https://pith.science/pith/O4GININTXG55VYYAJBJOLCKCQ7","download_json":"https://pith.science/pith/O4GININTXG55VYYAJBJOLCKCQ7.json","view_paper":"https://pith.science/paper/O4GININT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.11423&json=true","fetch_graph":"https://pith.science/api/pith-number/O4GININTXG55VYYAJBJOLCKCQ7/graph.json","fetch_events":"https://pith.science/api/pith-number/O4GININTXG55VYYAJBJOLCKCQ7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O4GININTXG55VYYAJBJOLCKCQ7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O4GININTXG55VYYAJBJOLCKCQ7/action/storage_attestation","attest_author":"https://pith.science/pith/O4GININTXG55VYYAJBJOLCKCQ7/action/author_attestation","sign_citation":"https://pith.science/pith/O4GININTXG55VYYAJBJOLCKCQ7/action/citation_signature","submit_replication":"https://pith.science/pith/O4GININTXG55VYYAJBJOLCKCQ7/action/replication_record"}},"created_at":"2026-07-05T10:06:43.854212+00:00","updated_at":"2026-07-05T10:06:43.854212+00:00"}