{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4USDAWUJ3SINWIPGCYUJ7JXIAI","short_pith_number":"pith:4USDAWUJ","schema_version":"1.0","canonical_sha256":"e524305a89dc90db21e616289fa6e8023c895f961bed68b0f0488cd22caff78e","source":{"kind":"arxiv","id":"2503.13772","version":1},"attestation_state":"computed","paper":{"title":"Do Large Language Models Understand Performance Optimization?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.DC","authors_text":"Bowen Cui, Keren Zhou, Oscar Hernandez, Tejas Ramesh","submitted_at":"2025-03-17T23:30:23Z","abstract_excerpt":"Large Language Models (LLMs) have emerged as powerful tools for software development tasks such as code completion, translation, and optimization. However, their ability to generate efficient and correct code, particularly in complex High-Performance Computing (HPC) contexts, has remained underexplored. To address this gap, this paper presents a comprehensive benchmark suite encompassing multiple critical HPC computational motifs to evaluate the performance of code optimized by state-of-the-art LLMs, including OpenAI o1, Claude-3.5, and Llama-3.2. In addition to analyzing basic computational k"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.13772","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-03-17T23:30:23Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"300d109db26eb7588e123b58174638c7a630ee7f2d1c0c96cc5138700f6ddc20","abstract_canon_sha256":"61264098ca61c709a25bce5f3568f0fcc8a780e6d4ddf6fb9404839a57e45700"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:33:51.268977Z","signature_b64":"ANp5kA/LjHXEtsEe6tJrj2YAE3PTt/G/+D0RVdfGIvlV3ytUVH7dkopgvgQW85VGyt+C1vAtclqpHsy6nN3FCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e524305a89dc90db21e616289fa6e8023c895f961bed68b0f0488cd22caff78e","last_reissued_at":"2026-07-05T10:33:51.268118Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:33:51.268118Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Large Language Models Understand Performance Optimization?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.DC","authors_text":"Bowen Cui, Keren Zhou, Oscar Hernandez, Tejas Ramesh","submitted_at":"2025-03-17T23:30:23Z","abstract_excerpt":"Large Language Models (LLMs) have emerged as powerful tools for software development tasks such as code completion, translation, and optimization. However, their ability to generate efficient and correct code, particularly in complex High-Performance Computing (HPC) contexts, has remained underexplored. To address this gap, this paper presents a comprehensive benchmark suite encompassing multiple critical HPC computational motifs to evaluate the performance of code optimized by state-of-the-art LLMs, including OpenAI o1, Claude-3.5, and Llama-3.2. In addition to analyzing basic computational k"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.13772","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.13772/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.13772","created_at":"2026-07-05T10:33:51.268223+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.13772v1","created_at":"2026-07-05T10:33:51.268223+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.13772","created_at":"2026-07-05T10:33:51.268223+00:00"},{"alias_kind":"pith_short_12","alias_value":"4USDAWUJ3SIN","created_at":"2026-07-05T10:33:51.268223+00:00"},{"alias_kind":"pith_short_16","alias_value":"4USDAWUJ3SINWIPG","created_at":"2026-07-05T10:33:51.268223+00:00"},{"alias_kind":"pith_short_8","alias_value":"4USDAWUJ","created_at":"2026-07-05T10:33:51.268223+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.16534","citing_title":"Generated, Parallel, Scalable? A Study of Agentic AI-Generated Julia Code on Supercomputers","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2505.13766","citing_title":"A Blueprint for AI-Driven Software Quality: Integrating LLMs with Established Standards","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20244","citing_title":"Lean Refactor: Multi-Objective Controllable Proof Optimization via Agentic Strategy Search","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04467","citing_title":"KEET: Explaining Performance of GPU Kernels Using LLM Agents","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24464","citing_title":"Incisor: Ex Ante Cloud Instance Selection for HPC Jobs","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4USDAWUJ3SINWIPGCYUJ7JXIAI","json":"https://pith.science/pith/4USDAWUJ3SINWIPGCYUJ7JXIAI.json","graph_json":"https://pith.science/api/pith-number/4USDAWUJ3SINWIPGCYUJ7JXIAI/graph.json","events_json":"https://pith.science/api/pith-number/4USDAWUJ3SINWIPGCYUJ7JXIAI/events.json","paper":"https://pith.science/paper/4USDAWUJ"},"agent_actions":{"view_html":"https://pith.science/pith/4USDAWUJ3SINWIPGCYUJ7JXIAI","download_json":"https://pith.science/pith/4USDAWUJ3SINWIPGCYUJ7JXIAI.json","view_paper":"https://pith.science/paper/4USDAWUJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.13772&json=true","fetch_graph":"https://pith.science/api/pith-number/4USDAWUJ3SINWIPGCYUJ7JXIAI/graph.json","fetch_events":"https://pith.science/api/pith-number/4USDAWUJ3SINWIPGCYUJ7JXIAI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4USDAWUJ3SINWIPGCYUJ7JXIAI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4USDAWUJ3SINWIPGCYUJ7JXIAI/action/storage_attestation","attest_author":"https://pith.science/pith/4USDAWUJ3SINWIPGCYUJ7JXIAI/action/author_attestation","sign_citation":"https://pith.science/pith/4USDAWUJ3SINWIPGCYUJ7JXIAI/action/citation_signature","submit_replication":"https://pith.science/pith/4USDAWUJ3SINWIPGCYUJ7JXIAI/action/replication_record"}},"created_at":"2026-07-05T10:33:51.268223+00:00","updated_at":"2026-07-05T10:33:51.268223+00:00"}