{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GCWLSTE3YQY6IFF6BOEILJ6TOJ","short_pith_number":"pith:GCWLSTE3","schema_version":"1.0","canonical_sha256":"30acb94c9bc431e414be0b8885a7d3725c3ce6b2e6a7919c0c866278d94706d2","source":{"kind":"arxiv","id":"2506.17330","version":1},"attestation_state":"computed","paper":{"title":"Large Language Models for Spreadsheets: Benchmarking Progress and Evaluating Performance with FLARE","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Simon Thorne","submitted_at":"2025-06-19T03:47:38Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated some significant capabilities across various domains; however, their effectiveness in spreadsheet related tasks remains underexplored. This study introduces a foundation for a comprehensive benchmark framework to evaluate the performance of leading LLMs in executing spreadsheet functions, formula generation and data manipulation tasks. The benchmark encompasses tasks ranging from basic formula creation to complex, real world spreadsheet scenarios. Our findings reveal that while LLMs exhibit proficiency in straightforward tasks, they often falter i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.17330","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2025-06-19T03:47:38Z","cross_cats_sorted":[],"title_canon_sha256":"a6f3c4f9c3a2ca232104d70f3fca709d02a8358421d8d278526fd57f7b18f8c1","abstract_canon_sha256":"76a4fc84bfac3dbc6c73aec3b608f373a01f662c0a21298605ed5f5b373dc6f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:16.002999Z","signature_b64":"7JuGWui+Ig1eTz7rd6nP2d0YScy5ea9tLdTUb6h5zmwo0LWtayAnjmTErRedvMRsH6xEfDbV5OfVVFjfvHmqDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"30acb94c9bc431e414be0b8885a7d3725c3ce6b2e6a7919c0c866278d94706d2","last_reissued_at":"2026-07-05T11:25:16.002495Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:16.002495Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models for Spreadsheets: Benchmarking Progress and Evaluating Performance with FLARE","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Simon Thorne","submitted_at":"2025-06-19T03:47:38Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated some significant capabilities across various domains; however, their effectiveness in spreadsheet related tasks remains underexplored. This study introduces a foundation for a comprehensive benchmark framework to evaluate the performance of leading LLMs in executing spreadsheet functions, formula generation and data manipulation tasks. The benchmark encompasses tasks ranging from basic formula creation to complex, real world spreadsheet scenarios. Our findings reveal that while LLMs exhibit proficiency in straightforward tasks, they often falter i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.17330","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.17330/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.17330","created_at":"2026-07-05T11:25:16.002557+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.17330v1","created_at":"2026-07-05T11:25:16.002557+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.17330","created_at":"2026-07-05T11:25:16.002557+00:00"},{"alias_kind":"pith_short_12","alias_value":"GCWLSTE3YQY6","created_at":"2026-07-05T11:25:16.002557+00:00"},{"alias_kind":"pith_short_16","alias_value":"GCWLSTE3YQY6IFF6","created_at":"2026-07-05T11:25:16.002557+00:00"},{"alias_kind":"pith_short_8","alias_value":"GCWLSTE3","created_at":"2026-07-05T11:25:16.002557+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29955","citing_title":"SpreadsheetBench 2: Evaluating Agents on End-to-End Business Spreadsheet Workflows","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05912","citing_title":"FrontierFinance: A Long-Horizon Computer-Use Benchmark of Real-World Financial Tasks","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GCWLSTE3YQY6IFF6BOEILJ6TOJ","json":"https://pith.science/pith/GCWLSTE3YQY6IFF6BOEILJ6TOJ.json","graph_json":"https://pith.science/api/pith-number/GCWLSTE3YQY6IFF6BOEILJ6TOJ/graph.json","events_json":"https://pith.science/api/pith-number/GCWLSTE3YQY6IFF6BOEILJ6TOJ/events.json","paper":"https://pith.science/paper/GCWLSTE3"},"agent_actions":{"view_html":"https://pith.science/pith/GCWLSTE3YQY6IFF6BOEILJ6TOJ","download_json":"https://pith.science/pith/GCWLSTE3YQY6IFF6BOEILJ6TOJ.json","view_paper":"https://pith.science/paper/GCWLSTE3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.17330&json=true","fetch_graph":"https://pith.science/api/pith-number/GCWLSTE3YQY6IFF6BOEILJ6TOJ/graph.json","fetch_events":"https://pith.science/api/pith-number/GCWLSTE3YQY6IFF6BOEILJ6TOJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GCWLSTE3YQY6IFF6BOEILJ6TOJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GCWLSTE3YQY6IFF6BOEILJ6TOJ/action/storage_attestation","attest_author":"https://pith.science/pith/GCWLSTE3YQY6IFF6BOEILJ6TOJ/action/author_attestation","sign_citation":"https://pith.science/pith/GCWLSTE3YQY6IFF6BOEILJ6TOJ/action/citation_signature","submit_replication":"https://pith.science/pith/GCWLSTE3YQY6IFF6BOEILJ6TOJ/action/replication_record"}},"created_at":"2026-07-05T11:25:16.002557+00:00","updated_at":"2026-07-05T11:25:16.002557+00:00"}