{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Q4NVFRZRV6VJBQYANLFKBBFMEH","short_pith_number":"pith:Q4NVFRZR","schema_version":"1.0","canonical_sha256":"871b52c731afaa90c3006acaa084ac21fed1f5eb671a31f315a7d9dd64a3b5dc","source":{"kind":"arxiv","id":"2412.11385","version":1},"attestation_state":"computed","paper":{"title":"Why Does ChatGPT \"Delve\" So Much? Exploring the Sources of Lexical Overrepresentation in Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Tom S. Juzek, Zina B. Ward","submitted_at":"2024-12-16T02:27:59Z","abstract_excerpt":"Scientific English is currently undergoing rapid change, with words like \"delve,\" \"intricate,\" and \"underscore\" appearing far more frequently than just a few years ago. It is widely assumed that scientists' use of large language models (LLMs) is responsible for such trends. We develop a formal, transferable method to characterize these linguistic changes. Application of our method yields 21 focal words whose increased occurrence in scientific abstracts is likely the result of LLM usage. We then pose \"the puzzle of lexical overrepresentation\": WHY are such words overused by LLMs? We fail to fin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.11385","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-16T02:27:59Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"663c6b5f3fa6b1a9e32ab4f3a8950dfa5cd926be0a3627a00643f7a1a83413cb","abstract_canon_sha256":"03950bae49c19b180c1ea6e38c50e82e4063d9a79bc50005a857941bdbfe4013"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:49:31.834598Z","signature_b64":"8qe5DxYH88XtzmuwaLtd3Gv/BjX102IFiGb1TtrpiNiQRfweiigFdSDFyl4VfFjGNuIczJU3PPOwmtWlgd2MDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"871b52c731afaa90c3006acaa084ac21fed1f5eb671a31f315a7d9dd64a3b5dc","last_reissued_at":"2026-07-05T09:49:31.834139Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:49:31.834139Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Why Does ChatGPT \"Delve\" So Much? Exploring the Sources of Lexical Overrepresentation in Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Tom S. Juzek, Zina B. Ward","submitted_at":"2024-12-16T02:27:59Z","abstract_excerpt":"Scientific English is currently undergoing rapid change, with words like \"delve,\" \"intricate,\" and \"underscore\" appearing far more frequently than just a few years ago. It is widely assumed that scientists' use of large language models (LLMs) is responsible for such trends. We develop a formal, transferable method to characterize these linguistic changes. Application of our method yields 21 focal words whose increased occurrence in scientific abstracts is likely the result of LLM usage. We then pose \"the puzzle of lexical overrepresentation\": WHY are such words overused by LLMs? We fail to fin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.11385","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.11385/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.11385","created_at":"2026-07-05T09:49:31.834210+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.11385v1","created_at":"2026-07-05T09:49:31.834210+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.11385","created_at":"2026-07-05T09:49:31.834210+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q4NVFRZRV6VJ","created_at":"2026-07-05T09:49:31.834210+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q4NVFRZRV6VJBQYA","created_at":"2026-07-05T09:49:31.834210+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q4NVFRZR","created_at":"2026-07-05T09:49:31.834210+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.01930","citing_title":"Word Overuse and Alignment in Large Language Models: The Influence of Learning from Human Feedback","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q4NVFRZRV6VJBQYANLFKBBFMEH","json":"https://pith.science/pith/Q4NVFRZRV6VJBQYANLFKBBFMEH.json","graph_json":"https://pith.science/api/pith-number/Q4NVFRZRV6VJBQYANLFKBBFMEH/graph.json","events_json":"https://pith.science/api/pith-number/Q4NVFRZRV6VJBQYANLFKBBFMEH/events.json","paper":"https://pith.science/paper/Q4NVFRZR"},"agent_actions":{"view_html":"https://pith.science/pith/Q4NVFRZRV6VJBQYANLFKBBFMEH","download_json":"https://pith.science/pith/Q4NVFRZRV6VJBQYANLFKBBFMEH.json","view_paper":"https://pith.science/paper/Q4NVFRZR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.11385&json=true","fetch_graph":"https://pith.science/api/pith-number/Q4NVFRZRV6VJBQYANLFKBBFMEH/graph.json","fetch_events":"https://pith.science/api/pith-number/Q4NVFRZRV6VJBQYANLFKBBFMEH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q4NVFRZRV6VJBQYANLFKBBFMEH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q4NVFRZRV6VJBQYANLFKBBFMEH/action/storage_attestation","attest_author":"https://pith.science/pith/Q4NVFRZRV6VJBQYANLFKBBFMEH/action/author_attestation","sign_citation":"https://pith.science/pith/Q4NVFRZRV6VJBQYANLFKBBFMEH/action/citation_signature","submit_replication":"https://pith.science/pith/Q4NVFRZRV6VJBQYANLFKBBFMEH/action/replication_record"}},"created_at":"2026-07-05T09:49:31.834210+00:00","updated_at":"2026-07-05T09:49:31.834210+00:00"}