{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7TEKAHEGSMCFIEQGHUYVYVE5NH","short_pith_number":"pith:7TEKAHEG","schema_version":"1.0","canonical_sha256":"fcc8a01c8693045412063d315c549d69ee3f501779f9ab84f74249f46351054b","source":{"kind":"arxiv","id":"2504.00025","version":1},"attestation_state":"computed","paper":{"title":"Generalization Bias in Large Language Model Summarization of Scientific Research","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Benjamin Chin-Yee, Uwe Peters","submitted_at":"2025-03-28T19:23:41Z","abstract_excerpt":"Artificial intelligence chatbots driven by large language models (LLMs) have the potential to increase public science literacy and support scientific research, as they can quickly summarize complex scientific information in accessible terms. However, when summarizing scientific texts, LLMs may omit details that limit the scope of research conclusions, leading to generalizations of results broader than warranted by the original study. We tested 10 prominent LLMs, including ChatGPT-4o, ChatGPT-4.5, DeepSeek, LLaMA 3.3 70B, and Claude 3.7 Sonnet, comparing 4900 LLM-generated summaries to their or"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.00025","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-28T19:23:41Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"80c2dbaa5b11c4e0cfee8f9d29a801059bdf5446e69290b7fa34262ebd2ae634","abstract_canon_sha256":"e7c691d342194aaad156bce7400c6d26c7e13a5553d449a2c55c3a7db56d1b75"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:42:18.580722Z","signature_b64":"On9oF3RUXps3c27jVHTNAgD8Pz9lfLVKwZQVIOTHC0l5mNUX7xHS+Hm0QVru0Mfm5jHU7ydoTag6zFURpqTGDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fcc8a01c8693045412063d315c549d69ee3f501779f9ab84f74249f46351054b","last_reissued_at":"2026-07-05T10:42:18.580256Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:42:18.580256Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalization Bias in Large Language Model Summarization of Scientific Research","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Benjamin Chin-Yee, Uwe Peters","submitted_at":"2025-03-28T19:23:41Z","abstract_excerpt":"Artificial intelligence chatbots driven by large language models (LLMs) have the potential to increase public science literacy and support scientific research, as they can quickly summarize complex scientific information in accessible terms. However, when summarizing scientific texts, LLMs may omit details that limit the scope of research conclusions, leading to generalizations of results broader than warranted by the original study. We tested 10 prominent LLMs, including ChatGPT-4o, ChatGPT-4.5, DeepSeek, LLaMA 3.3 70B, and Claude 3.7 Sonnet, comparing 4900 LLM-generated summaries to their or"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.00025","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.00025/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.00025","created_at":"2026-07-05T10:42:18.580312+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.00025v1","created_at":"2026-07-05T10:42:18.580312+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.00025","created_at":"2026-07-05T10:42:18.580312+00:00"},{"alias_kind":"pith_short_12","alias_value":"7TEKAHEGSMCF","created_at":"2026-07-05T10:42:18.580312+00:00"},{"alias_kind":"pith_short_16","alias_value":"7TEKAHEGSMCFIEQG","created_at":"2026-07-05T10:42:18.580312+00:00"},{"alias_kind":"pith_short_8","alias_value":"7TEKAHEG","created_at":"2026-07-05T10:42:18.580312+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29251","citing_title":"When Summaries Distort Decisions: Information Fidelity in LLM-Compressed Financial Analysis","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7TEKAHEGSMCFIEQGHUYVYVE5NH","json":"https://pith.science/pith/7TEKAHEGSMCFIEQGHUYVYVE5NH.json","graph_json":"https://pith.science/api/pith-number/7TEKAHEGSMCFIEQGHUYVYVE5NH/graph.json","events_json":"https://pith.science/api/pith-number/7TEKAHEGSMCFIEQGHUYVYVE5NH/events.json","paper":"https://pith.science/paper/7TEKAHEG"},"agent_actions":{"view_html":"https://pith.science/pith/7TEKAHEGSMCFIEQGHUYVYVE5NH","download_json":"https://pith.science/pith/7TEKAHEGSMCFIEQGHUYVYVE5NH.json","view_paper":"https://pith.science/paper/7TEKAHEG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.00025&json=true","fetch_graph":"https://pith.science/api/pith-number/7TEKAHEGSMCFIEQGHUYVYVE5NH/graph.json","fetch_events":"https://pith.science/api/pith-number/7TEKAHEGSMCFIEQGHUYVYVE5NH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7TEKAHEGSMCFIEQGHUYVYVE5NH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7TEKAHEGSMCFIEQGHUYVYVE5NH/action/storage_attestation","attest_author":"https://pith.science/pith/7TEKAHEGSMCFIEQGHUYVYVE5NH/action/author_attestation","sign_citation":"https://pith.science/pith/7TEKAHEGSMCFIEQGHUYVYVE5NH/action/citation_signature","submit_replication":"https://pith.science/pith/7TEKAHEGSMCFIEQGHUYVYVE5NH/action/replication_record"}},"created_at":"2026-07-05T10:42:18.580312+00:00","updated_at":"2026-07-05T10:42:18.580312+00:00"}