{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZLFX3EBKNR2AMLBVUCHYWKMZFL","short_pith_number":"pith:ZLFX3EBK","schema_version":"1.0","canonical_sha256":"cacb7d902a6c74062c35a08f8b29992ada4719c478cb5f0870938b8bcb0c5426","source":{"kind":"arxiv","id":"2404.17143","version":2},"attestation_state":"computed","paper":{"title":"Quantifying Memorization and Detecting Training Data of Pre-trained Language Models using Japanese Newspaper","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hiromu Takahashi, Shotaro Ishihara","submitted_at":"2024-04-26T04:12:08Z","abstract_excerpt":"Dominant pre-trained language models (PLMs) have demonstrated the potential risk of memorizing and outputting the training data. While this concern has been discussed mainly in English, it is also practically important to focus on domain-specific PLMs. In this study, we pre-trained domain-specific GPT-2 models using a limited corpus of Japanese newspaper articles and evaluated their behavior. Experiments replicated the empirical finding that memorization of PLMs is related to the duplication in the training data, model size, and prompt length, in Japanese the same as in previous English studie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.17143","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-26T04:12:08Z","cross_cats_sorted":[],"title_canon_sha256":"0e5e810a783eb6385681a5821e59a27bf2dcaebee7bad15696109f1e795517f0","abstract_canon_sha256":"4d81e07a85f57b536114bbd124f0f5b2691f74dc860f3b7f7462803e289e46cc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:55:35.504030Z","signature_b64":"SNOxeOnrggZvRdK8XKoIPJJjljcBRpSpF/Ccym7kU2+oDZYZS0TAK3WYAJZG74uQ9FGk2a66OfL2LvPWJp95CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cacb7d902a6c74062c35a08f8b29992ada4719c478cb5f0870938b8bcb0c5426","last_reissued_at":"2026-07-05T08:55:35.503539Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:55:35.503539Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quantifying Memorization and Detecting Training Data of Pre-trained Language Models using Japanese Newspaper","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hiromu Takahashi, Shotaro Ishihara","submitted_at":"2024-04-26T04:12:08Z","abstract_excerpt":"Dominant pre-trained language models (PLMs) have demonstrated the potential risk of memorizing and outputting the training data. While this concern has been discussed mainly in English, it is also practically important to focus on domain-specific PLMs. In this study, we pre-trained domain-specific GPT-2 models using a limited corpus of Japanese newspaper articles and evaluated their behavior. Experiments replicated the empirical finding that memorization of PLMs is related to the duplication in the training data, model size, and prompt length, in Japanese the same as in previous English studie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.17143","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.17143/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.17143","created_at":"2026-07-05T08:55:35.503598+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.17143v2","created_at":"2026-07-05T08:55:35.503598+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.17143","created_at":"2026-07-05T08:55:35.503598+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZLFX3EBKNR2A","created_at":"2026-07-05T08:55:35.503598+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZLFX3EBKNR2AMLBV","created_at":"2026-07-05T08:55:35.503598+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZLFX3EBK","created_at":"2026-07-05T08:55:35.503598+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.05078","citing_title":"Analyzing Memorization in Large Language Models through the Lens of Model Attribution","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZLFX3EBKNR2AMLBVUCHYWKMZFL","json":"https://pith.science/pith/ZLFX3EBKNR2AMLBVUCHYWKMZFL.json","graph_json":"https://pith.science/api/pith-number/ZLFX3EBKNR2AMLBVUCHYWKMZFL/graph.json","events_json":"https://pith.science/api/pith-number/ZLFX3EBKNR2AMLBVUCHYWKMZFL/events.json","paper":"https://pith.science/paper/ZLFX3EBK"},"agent_actions":{"view_html":"https://pith.science/pith/ZLFX3EBKNR2AMLBVUCHYWKMZFL","download_json":"https://pith.science/pith/ZLFX3EBKNR2AMLBVUCHYWKMZFL.json","view_paper":"https://pith.science/paper/ZLFX3EBK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.17143&json=true","fetch_graph":"https://pith.science/api/pith-number/ZLFX3EBKNR2AMLBVUCHYWKMZFL/graph.json","fetch_events":"https://pith.science/api/pith-number/ZLFX3EBKNR2AMLBVUCHYWKMZFL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZLFX3EBKNR2AMLBVUCHYWKMZFL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZLFX3EBKNR2AMLBVUCHYWKMZFL/action/storage_attestation","attest_author":"https://pith.science/pith/ZLFX3EBKNR2AMLBVUCHYWKMZFL/action/author_attestation","sign_citation":"https://pith.science/pith/ZLFX3EBKNR2AMLBVUCHYWKMZFL/action/citation_signature","submit_replication":"https://pith.science/pith/ZLFX3EBKNR2AMLBVUCHYWKMZFL/action/replication_record"}},"created_at":"2026-07-05T08:55:35.503598+00:00","updated_at":"2026-07-05T08:55:35.503598+00:00"}