{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VR6ZIPMZRO6CEHP7VU3HCVVGNC","short_pith_number":"pith:VR6ZIPMZ","schema_version":"1.0","canonical_sha256":"ac7d943d998bbc221dffad367156a6688cb9d9a02417d86ce544874475119ad8","source":{"kind":"arxiv","id":"2504.02767","version":1},"attestation_state":"computed","paper":{"title":"How Deep Do Large Language Models Internalize Scientific Literature and Citation Practices?","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SI"],"primary_cat":"cs.DL","authors_text":"Andres Algaba, Brecht Verbeken, Floriano Tori, Melika Mobini, Sylvia Wenmackers, Vincent Ginis, Vincent Holst","submitted_at":"2025-04-03T17:04:56Z","abstract_excerpt":"The spread of scientific knowledge depends on how researchers discover and cite previous work. The adoption of large language models (LLMs) in the scientific research process introduces a new layer to these citation practices. However, it remains unclear to what extent LLMs align with human citation practices, how they perform across domains, and may influence citation dynamics. Here, we show that LLMs systematically reinforce the Matthew effect in citations by consistently favoring highly cited papers when generating references. This pattern persists across scientific domains despite signific"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.02767","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.DL","submitted_at":"2025-04-03T17:04:56Z","cross_cats_sorted":["cs.AI","cs.LG","cs.SI"],"title_canon_sha256":"743fc7d77768f944004f317ae7ff3848f482b8ee2c47a460282b21b053dc36ff","abstract_canon_sha256":"0d1bfd8d54130353af6d7de49b5c276ef6760c63ad06d4394a85f16d46017825"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:44:02.023821Z","signature_b64":"YknObYnLEdflWto+hxbX3gwFAsp75bqS4nvcjYNvxNE0RVG4dyxzfHhI2EuoYXcBkaynb5hQrf4nMsXccj82Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac7d943d998bbc221dffad367156a6688cb9d9a02417d86ce544874475119ad8","last_reissued_at":"2026-07-05T10:44:02.023334Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:44:02.023334Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Deep Do Large Language Models Internalize Scientific Literature and Citation Practices?","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SI"],"primary_cat":"cs.DL","authors_text":"Andres Algaba, Brecht Verbeken, Floriano Tori, Melika Mobini, Sylvia Wenmackers, Vincent Ginis, Vincent Holst","submitted_at":"2025-04-03T17:04:56Z","abstract_excerpt":"The spread of scientific knowledge depends on how researchers discover and cite previous work. The adoption of large language models (LLMs) in the scientific research process introduces a new layer to these citation practices. However, it remains unclear to what extent LLMs align with human citation practices, how they perform across domains, and may influence citation dynamics. Here, we show that LLMs systematically reinforce the Matthew effect in citations by consistently favoring highly cited papers when generating references. This pattern persists across scientific domains despite signific"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.02767","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.02767/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.02767","created_at":"2026-07-05T10:44:02.023398+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.02767v1","created_at":"2026-07-05T10:44:02.023398+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.02767","created_at":"2026-07-05T10:44:02.023398+00:00"},{"alias_kind":"pith_short_12","alias_value":"VR6ZIPMZRO6C","created_at":"2026-07-05T10:44:02.023398+00:00"},{"alias_kind":"pith_short_16","alias_value":"VR6ZIPMZRO6CEHP7","created_at":"2026-07-05T10:44:02.023398+00:00"},{"alias_kind":"pith_short_8","alias_value":"VR6ZIPMZ","created_at":"2026-07-05T10:44:02.023398+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20065","citing_title":"Generative Engine Optimization at Scale: Measuring Brand Visibility Across AI Search Engines","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26130","citing_title":"Thinking Like a Scientist? A Structural Study of LLM-Generated Research Methods","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VR6ZIPMZRO6CEHP7VU3HCVVGNC","json":"https://pith.science/pith/VR6ZIPMZRO6CEHP7VU3HCVVGNC.json","graph_json":"https://pith.science/api/pith-number/VR6ZIPMZRO6CEHP7VU3HCVVGNC/graph.json","events_json":"https://pith.science/api/pith-number/VR6ZIPMZRO6CEHP7VU3HCVVGNC/events.json","paper":"https://pith.science/paper/VR6ZIPMZ"},"agent_actions":{"view_html":"https://pith.science/pith/VR6ZIPMZRO6CEHP7VU3HCVVGNC","download_json":"https://pith.science/pith/VR6ZIPMZRO6CEHP7VU3HCVVGNC.json","view_paper":"https://pith.science/paper/VR6ZIPMZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.02767&json=true","fetch_graph":"https://pith.science/api/pith-number/VR6ZIPMZRO6CEHP7VU3HCVVGNC/graph.json","fetch_events":"https://pith.science/api/pith-number/VR6ZIPMZRO6CEHP7VU3HCVVGNC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VR6ZIPMZRO6CEHP7VU3HCVVGNC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VR6ZIPMZRO6CEHP7VU3HCVVGNC/action/storage_attestation","attest_author":"https://pith.science/pith/VR6ZIPMZRO6CEHP7VU3HCVVGNC/action/author_attestation","sign_citation":"https://pith.science/pith/VR6ZIPMZRO6CEHP7VU3HCVVGNC/action/citation_signature","submit_replication":"https://pith.science/pith/VR6ZIPMZRO6CEHP7VU3HCVVGNC/action/replication_record"}},"created_at":"2026-07-05T10:44:02.023398+00:00","updated_at":"2026-07-05T10:44:02.023398+00:00"}