{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BVXXRL6H2IS5ES7ZKH4KOVGTNC","short_pith_number":"pith:BVXXRL6H","schema_version":"1.0","canonical_sha256":"0d6f78afc7d225d24bf951f8a754d368bf7e0ea5b4931b2a1a486f6f2fb9be03","source":{"kind":"arxiv","id":"2504.10415","version":2},"attestation_state":"computed","paper":{"title":"LLM-SRBench: A New Benchmark for Scientific Equation Discovery with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Amir Barati Farimani, Chandan K Reddy, Kazem Meidani, Khoa D Doan, Ngoc-Hieu Nguyen, Parshin Shojaee","submitted_at":"2025-04-14T17:00:13Z","abstract_excerpt":"Scientific equation discovery is a fundamental task in the history of scientific progress, enabling the derivation of laws governing natural phenomena. Recently, Large Language Models (LLMs) have gained interest for this task due to their potential to leverage embedded scientific knowledge for hypothesis generation. However, evaluating the true discovery capabilities of these methods remains challenging, as existing benchmarks often rely on common equations that are susceptible to memorization by LLMs, leading to inflated performance metrics that do not reflect discovery. In this paper, we int"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.10415","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-14T17:00:13Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b3bd2c10029af5d6ea8ed7698d5ac5d63b1c32fa66721c64a7c81759ee1c73aa","abstract_canon_sha256":"9dde30b62b895d07581e31371a48bfa04de2941403c918d2615ed333df017bd7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:44.755506Z","signature_b64":"GIolmGCWQsK2TK7Ez2nk+gywpp/CcWlEsFsh00IY0kuKyOLjfAl1FWDVaTR15wkQ+4V75R5qOlO6LnL6lHY7AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0d6f78afc7d225d24bf951f8a754d368bf7e0ea5b4931b2a1a486f6f2fb9be03","last_reissued_at":"2026-07-05T11:17:44.755020Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:44.755020Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM-SRBench: A New Benchmark for Scientific Equation Discovery with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Amir Barati Farimani, Chandan K Reddy, Kazem Meidani, Khoa D Doan, Ngoc-Hieu Nguyen, Parshin Shojaee","submitted_at":"2025-04-14T17:00:13Z","abstract_excerpt":"Scientific equation discovery is a fundamental task in the history of scientific progress, enabling the derivation of laws governing natural phenomena. Recently, Large Language Models (LLMs) have gained interest for this task due to their potential to leverage embedded scientific knowledge for hypothesis generation. However, evaluating the true discovery capabilities of these methods remains challenging, as existing benchmarks often rely on common equations that are susceptible to memorization by LLMs, leading to inflated performance metrics that do not reflect discovery. In this paper, we int"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.10415","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.10415/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.10415","created_at":"2026-07-05T11:17:44.755078+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.10415v2","created_at":"2026-07-05T11:17:44.755078+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.10415","created_at":"2026-07-05T11:17:44.755078+00:00"},{"alias_kind":"pith_short_12","alias_value":"BVXXRL6H2IS5","created_at":"2026-07-05T11:17:44.755078+00:00"},{"alias_kind":"pith_short_16","alias_value":"BVXXRL6H2IS5ES7Z","created_at":"2026-07-05T11:17:44.755078+00:00"},{"alias_kind":"pith_short_8","alias_value":"BVXXRL6H","created_at":"2026-07-05T11:17:44.755078+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10587","citing_title":"Towards Diverse Scientific Hypothesis Search with Large Language Models","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07915","citing_title":"EditSR: Enhancing Neural Symbolic Regression via Edit-based Rectification","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07704","citing_title":"FunctionEvolve: Structure-Guided Symbolic Regression with LLMs","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00733","citing_title":"LLM-Guided ODE Discovery and Parameter Inference from Small-Cohort Aggregate Data","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01286","citing_title":"BenchEvolver: Frontier Task Synthesis via Solution-Centric Evolution","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29082","citing_title":"Evolution Fine-Tuning: Learning to Discover Across 371 Optimization Tasks","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15308","citing_title":"SMCEvolve: Principled Scientific Discovery via Sequential Monte Carlo Evolution","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20910","citing_title":"LLM-ODE: Data-driven Discovery of Dynamical Systems with Large Language Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19341","citing_title":"Evaluation-driven Scaling for Scientific Discovery","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04854","citing_title":"Assessing Large Language Models for Stabilizing Numerical Expressions in Scientific Software","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03101","citing_title":"Programmatic Context Augmentation for LLM-based Symbolic Regression","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BVXXRL6H2IS5ES7ZKH4KOVGTNC","json":"https://pith.science/pith/BVXXRL6H2IS5ES7ZKH4KOVGTNC.json","graph_json":"https://pith.science/api/pith-number/BVXXRL6H2IS5ES7ZKH4KOVGTNC/graph.json","events_json":"https://pith.science/api/pith-number/BVXXRL6H2IS5ES7ZKH4KOVGTNC/events.json","paper":"https://pith.science/paper/BVXXRL6H"},"agent_actions":{"view_html":"https://pith.science/pith/BVXXRL6H2IS5ES7ZKH4KOVGTNC","download_json":"https://pith.science/pith/BVXXRL6H2IS5ES7ZKH4KOVGTNC.json","view_paper":"https://pith.science/paper/BVXXRL6H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.10415&json=true","fetch_graph":"https://pith.science/api/pith-number/BVXXRL6H2IS5ES7ZKH4KOVGTNC/graph.json","fetch_events":"https://pith.science/api/pith-number/BVXXRL6H2IS5ES7ZKH4KOVGTNC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BVXXRL6H2IS5ES7ZKH4KOVGTNC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BVXXRL6H2IS5ES7ZKH4KOVGTNC/action/storage_attestation","attest_author":"https://pith.science/pith/BVXXRL6H2IS5ES7ZKH4KOVGTNC/action/author_attestation","sign_citation":"https://pith.science/pith/BVXXRL6H2IS5ES7ZKH4KOVGTNC/action/citation_signature","submit_replication":"https://pith.science/pith/BVXXRL6H2IS5ES7ZKH4KOVGTNC/action/replication_record"}},"created_at":"2026-07-05T11:17:44.755078+00:00","updated_at":"2026-07-05T11:17:44.755078+00:00"}