{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DO2DVI67SJ6Q73DPUCLUPBOMZS","short_pith_number":"pith:DO2DVI67","schema_version":"1.0","canonical_sha256":"1bb43aa3df927d0fec6fa0974785cccc871056f684036a233d6d8c01bd284257","source":{"kind":"arxiv","id":"2508.01930","version":1},"attestation_state":"computed","paper":{"title":"Word Overuse and Alignment in Large Language Models: The Influence of Learning from Human Feedback","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Tom S. Juzek, Zina B. Ward","submitted_at":"2025-08-03T21:45:37Z","abstract_excerpt":"Large Language Models (LLMs) are known to overuse certain terms like \"delve\" and \"intricate.\" The exact reasons for these lexical choices, however, have been unclear. Using Meta's Llama model, this study investigates the contribution of Learning from Human Feedback (LHF), under which we subsume Reinforcement Learning from Human Feedback and Direct Preference Optimization. We present a straightforward procedure for detecting the lexical preferences of LLMs that are potentially LHF-induced. Next, we more conclusively link LHF to lexical overuse by experimentally emulating the LHF procedure and d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.01930","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-08-03T21:45:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2e6e4278004095999d5c722a1d535382b593925007b029aec2377b54315fd9ba","abstract_canon_sha256":"53b4f2afc973993e0a8dffa192fe0f3ce358e388ffc7b7b7ef7b3e30b5db9b10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:47:48.376396Z","signature_b64":"dm0FBBlHpspctECvDzyOp5/ouar+MIpv7a9WQj9SmBuuYQHLxhqk86ULaFUfQbrmItQB+WcPzsXJAIw3rOavBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1bb43aa3df927d0fec6fa0974785cccc871056f684036a233d6d8c01bd284257","last_reissued_at":"2026-07-05T11:47:48.375910Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:47:48.375910Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Word Overuse and Alignment in Large Language Models: The Influence of Learning from Human Feedback","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Tom S. Juzek, Zina B. Ward","submitted_at":"2025-08-03T21:45:37Z","abstract_excerpt":"Large Language Models (LLMs) are known to overuse certain terms like \"delve\" and \"intricate.\" The exact reasons for these lexical choices, however, have been unclear. Using Meta's Llama model, this study investigates the contribution of Learning from Human Feedback (LHF), under which we subsume Reinforcement Learning from Human Feedback and Direct Preference Optimization. We present a straightforward procedure for detecting the lexical preferences of LLMs that are potentially LHF-induced. Next, we more conclusively link LHF to lexical overuse by experimentally emulating the LHF procedure and d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.01930","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.01930/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.01930","created_at":"2026-07-05T11:47:48.375972+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.01930v1","created_at":"2026-07-05T11:47:48.375972+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.01930","created_at":"2026-07-05T11:47:48.375972+00:00"},{"alias_kind":"pith_short_12","alias_value":"DO2DVI67SJ6Q","created_at":"2026-07-05T11:47:48.375972+00:00"},{"alias_kind":"pith_short_16","alias_value":"DO2DVI67SJ6Q73DP","created_at":"2026-07-05T11:47:48.375972+00:00"},{"alias_kind":"pith_short_8","alias_value":"DO2DVI67","created_at":"2026-07-05T11:47:48.375972+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00334","citing_title":"Isolating LLM Lexical Bias: A Curation-Free Triangulated Metric for Preference-Stage Learning","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DO2DVI67SJ6Q73DPUCLUPBOMZS","json":"https://pith.science/pith/DO2DVI67SJ6Q73DPUCLUPBOMZS.json","graph_json":"https://pith.science/api/pith-number/DO2DVI67SJ6Q73DPUCLUPBOMZS/graph.json","events_json":"https://pith.science/api/pith-number/DO2DVI67SJ6Q73DPUCLUPBOMZS/events.json","paper":"https://pith.science/paper/DO2DVI67"},"agent_actions":{"view_html":"https://pith.science/pith/DO2DVI67SJ6Q73DPUCLUPBOMZS","download_json":"https://pith.science/pith/DO2DVI67SJ6Q73DPUCLUPBOMZS.json","view_paper":"https://pith.science/paper/DO2DVI67","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.01930&json=true","fetch_graph":"https://pith.science/api/pith-number/DO2DVI67SJ6Q73DPUCLUPBOMZS/graph.json","fetch_events":"https://pith.science/api/pith-number/DO2DVI67SJ6Q73DPUCLUPBOMZS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DO2DVI67SJ6Q73DPUCLUPBOMZS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DO2DVI67SJ6Q73DPUCLUPBOMZS/action/storage_attestation","attest_author":"https://pith.science/pith/DO2DVI67SJ6Q73DPUCLUPBOMZS/action/author_attestation","sign_citation":"https://pith.science/pith/DO2DVI67SJ6Q73DPUCLUPBOMZS/action/citation_signature","submit_replication":"https://pith.science/pith/DO2DVI67SJ6Q73DPUCLUPBOMZS/action/replication_record"}},"created_at":"2026-07-05T11:47:48.375972+00:00","updated_at":"2026-07-05T11:47:48.375972+00:00"}