{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RQCPP6WIGDI57LXXSVBI5DQPLK","short_pith_number":"pith:RQCPP6WI","schema_version":"1.0","canonical_sha256":"8c04f7fac830d1dfaef795428e8e0f5aa387c83fa60f8c95a325d60407eee6d4","source":{"kind":"arxiv","id":"2410.16410","version":1},"attestation_state":"computed","paper":{"title":"Subword Embedding from Bytes Gains Privacy without Sacrificing Accuracy and Complexity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Jia Xu, Mengjiao Zhang","submitted_at":"2024-10-21T18:25:24Z","abstract_excerpt":"While NLP models significantly impact our lives, there are rising concerns about privacy invasion. Although federated learning enhances privacy, attackers may recover private training data by exploiting model parameters and gradients. Therefore, protecting against such embedding attacks remains an open challenge. To address this, we propose Subword Embedding from Bytes (SEB) and encode subwords to byte sequences using deep neural networks, making input text recovery harder. Importantly, our method requires a smaller memory with $256$ bytes of vocabulary while keeping efficiency with the same i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.16410","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-10-21T18:25:24Z","cross_cats_sorted":[],"title_canon_sha256":"e34cd13ba7103e4ba5e7571324e9e907d1a8246d87bcd2d807011bb174c5d2a5","abstract_canon_sha256":"e522417ebcb47690bcd839b3b3f62b7e4ed900ef2487aab63a72cbff0c8400cb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:23:35.915899Z","signature_b64":"+1BfweX/u8dmjlIAL2mE64I2dy7lwnOiV8RVnWiDIzbb55E5yXdB3x8tQDlo0a2YzKNotek9rVooEgj8VrDRDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c04f7fac830d1dfaef795428e8e0f5aa387c83fa60f8c95a325d60407eee6d4","last_reissued_at":"2026-07-05T09:23:35.915453Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:23:35.915453Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Subword Embedding from Bytes Gains Privacy without Sacrificing Accuracy and Complexity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Jia Xu, Mengjiao Zhang","submitted_at":"2024-10-21T18:25:24Z","abstract_excerpt":"While NLP models significantly impact our lives, there are rising concerns about privacy invasion. Although federated learning enhances privacy, attackers may recover private training data by exploiting model parameters and gradients. Therefore, protecting against such embedding attacks remains an open challenge. To address this, we propose Subword Embedding from Bytes (SEB) and encode subwords to byte sequences using deep neural networks, making input text recovery harder. Importantly, our method requires a smaller memory with $256$ bytes of vocabulary while keeping efficiency with the same i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.16410","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.16410/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.16410","created_at":"2026-07-05T09:23:35.915505+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.16410v1","created_at":"2026-07-05T09:23:35.915505+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.16410","created_at":"2026-07-05T09:23:35.915505+00:00"},{"alias_kind":"pith_short_12","alias_value":"RQCPP6WIGDI5","created_at":"2026-07-05T09:23:35.915505+00:00"},{"alias_kind":"pith_short_16","alias_value":"RQCPP6WIGDI57LXX","created_at":"2026-07-05T09:23:35.915505+00:00"},{"alias_kind":"pith_short_8","alias_value":"RQCPP6WI","created_at":"2026-07-05T09:23:35.915505+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RQCPP6WIGDI57LXXSVBI5DQPLK","json":"https://pith.science/pith/RQCPP6WIGDI57LXXSVBI5DQPLK.json","graph_json":"https://pith.science/api/pith-number/RQCPP6WIGDI57LXXSVBI5DQPLK/graph.json","events_json":"https://pith.science/api/pith-number/RQCPP6WIGDI57LXXSVBI5DQPLK/events.json","paper":"https://pith.science/paper/RQCPP6WI"},"agent_actions":{"view_html":"https://pith.science/pith/RQCPP6WIGDI57LXXSVBI5DQPLK","download_json":"https://pith.science/pith/RQCPP6WIGDI57LXXSVBI5DQPLK.json","view_paper":"https://pith.science/paper/RQCPP6WI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.16410&json=true","fetch_graph":"https://pith.science/api/pith-number/RQCPP6WIGDI57LXXSVBI5DQPLK/graph.json","fetch_events":"https://pith.science/api/pith-number/RQCPP6WIGDI57LXXSVBI5DQPLK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RQCPP6WIGDI57LXXSVBI5DQPLK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RQCPP6WIGDI57LXXSVBI5DQPLK/action/storage_attestation","attest_author":"https://pith.science/pith/RQCPP6WIGDI57LXXSVBI5DQPLK/action/author_attestation","sign_citation":"https://pith.science/pith/RQCPP6WIGDI57LXXSVBI5DQPLK/action/citation_signature","submit_replication":"https://pith.science/pith/RQCPP6WIGDI57LXXSVBI5DQPLK/action/replication_record"}},"created_at":"2026-07-05T09:23:35.915505+00:00","updated_at":"2026-07-05T09:23:35.915505+00:00"}