{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2WPHHRMSBQWYJT6YYRTZ4TGL5A","short_pith_number":"pith:2WPHHRMS","schema_version":"1.0","canonical_sha256":"d59e73c5920c2d84cfd8c4679e4ccbe8359de56a4ec51daeb1762118d49a16ad","source":{"kind":"arxiv","id":"2406.10833","version":3},"attestation_state":"computed","paper":{"title":"A Comprehensive Survey of Scientific Large Language Models and Their Applications in Scientific Discovery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bowen Jin, Jiawei Han, Sheng Wang, Shuiwang Ji, Wei Wang, Xiusi Chen, Yu Zhang","submitted_at":"2024-06-16T08:03:24Z","abstract_excerpt":"In many scientific fields, large language models (LLMs) have revolutionized the way text and other modalities of data (e.g., molecules and proteins) are handled, achieving superior performance in various applications and augmenting the scientific discovery process. Nevertheless, previous surveys on scientific LLMs often concentrate on one or two fields or a single modality. In this paper, we aim to provide a more holistic view of the research landscape by unveiling cross-field and cross-modal connections between scientific LLMs regarding their architectures and pre-training techniques. To this"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10833","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-16T08:03:24Z","cross_cats_sorted":[],"title_canon_sha256":"309a6864f39ed185030b15b21e30fdf0ee510d8b613901bf4f058f145ee16f0a","abstract_canon_sha256":"f975153393d0778d82cfe497b4368168557342d08ea7a5d890d42d23cb60c181"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:03.696899Z","signature_b64":"95WNFlt8yBo2jNo9I+ZBQd6t8Dh8Rawrcf9Ujrktto0PPPH/kXvqbY0KyRjJurDfXSfHNMMqHnwG85+HPeXGBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d59e73c5920c2d84cfd8c4679e4ccbe8359de56a4ec51daeb1762118d49a16ad","last_reissued_at":"2026-07-05T09:13:03.696497Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:03.696497Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comprehensive Survey of Scientific Large Language Models and Their Applications in Scientific Discovery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bowen Jin, Jiawei Han, Sheng Wang, Shuiwang Ji, Wei Wang, Xiusi Chen, Yu Zhang","submitted_at":"2024-06-16T08:03:24Z","abstract_excerpt":"In many scientific fields, large language models (LLMs) have revolutionized the way text and other modalities of data (e.g., molecules and proteins) are handled, achieving superior performance in various applications and augmenting the scientific discovery process. Nevertheless, previous surveys on scientific LLMs often concentrate on one or two fields or a single modality. In this paper, we aim to provide a more holistic view of the research landscape by unveiling cross-field and cross-modal connections between scientific LLMs regarding their architectures and pre-training techniques. To this"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10833","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10833/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10833","created_at":"2026-07-05T09:13:03.696553+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10833v3","created_at":"2026-07-05T09:13:03.696553+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10833","created_at":"2026-07-05T09:13:03.696553+00:00"},{"alias_kind":"pith_short_12","alias_value":"2WPHHRMSBQWY","created_at":"2026-07-05T09:13:03.696553+00:00"},{"alias_kind":"pith_short_16","alias_value":"2WPHHRMSBQWYJT6Y","created_at":"2026-07-05T09:13:03.696553+00:00"},{"alias_kind":"pith_short_8","alias_value":"2WPHHRMS","created_at":"2026-07-05T09:13:03.696553+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":252,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21035","citing_title":"GenoMAS: A Multi-Agent Framework for Scientific Discovery via Code-Driven Gene Expression Analysis","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2WPHHRMSBQWYJT6YYRTZ4TGL5A","json":"https://pith.science/pith/2WPHHRMSBQWYJT6YYRTZ4TGL5A.json","graph_json":"https://pith.science/api/pith-number/2WPHHRMSBQWYJT6YYRTZ4TGL5A/graph.json","events_json":"https://pith.science/api/pith-number/2WPHHRMSBQWYJT6YYRTZ4TGL5A/events.json","paper":"https://pith.science/paper/2WPHHRMS"},"agent_actions":{"view_html":"https://pith.science/pith/2WPHHRMSBQWYJT6YYRTZ4TGL5A","download_json":"https://pith.science/pith/2WPHHRMSBQWYJT6YYRTZ4TGL5A.json","view_paper":"https://pith.science/paper/2WPHHRMS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10833&json=true","fetch_graph":"https://pith.science/api/pith-number/2WPHHRMSBQWYJT6YYRTZ4TGL5A/graph.json","fetch_events":"https://pith.science/api/pith-number/2WPHHRMSBQWYJT6YYRTZ4TGL5A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2WPHHRMSBQWYJT6YYRTZ4TGL5A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2WPHHRMSBQWYJT6YYRTZ4TGL5A/action/storage_attestation","attest_author":"https://pith.science/pith/2WPHHRMSBQWYJT6YYRTZ4TGL5A/action/author_attestation","sign_citation":"https://pith.science/pith/2WPHHRMSBQWYJT6YYRTZ4TGL5A/action/citation_signature","submit_replication":"https://pith.science/pith/2WPHHRMSBQWYJT6YYRTZ4TGL5A/action/replication_record"}},"created_at":"2026-07-05T09:13:03.696553+00:00","updated_at":"2026-07-05T09:13:03.696553+00:00"}