{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5K24PZ4NDYQHPI2EKZLN55DXBI","short_pith_number":"pith:5K24PZ4N","schema_version":"1.0","canonical_sha256":"eab5c7e78d1e2077a3445656def4770a0bd479363c10563cc795d9e908785a78","source":{"kind":"arxiv","id":"2507.14403","version":1},"attestation_state":"computed","paper":{"title":"NPUEval: Optimizing NPU Kernels with LLMs and Open Source Compilers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.PL","authors_text":"Graham Schelle, Sarunas Kalade","submitted_at":"2025-07-18T23:21:52Z","abstract_excerpt":"Neural processing units (NPUs) are gaining prominence in power-sensitive devices like client devices, with AI PCs being defined by their inclusion of these specialized processors. Running AI workloads efficiently on these devices requires libraries of optimized kernels. Creating efficient kernels demands expertise in domain-specific C++ with vector intrinsics and in-depth knowledge of the target architecture. Unlike GPU programming, which has had years to mature, NPU programming is new, with smaller and more fragmented developer communities across hardware platforms. This fragmentation poses a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.14403","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.PL","submitted_at":"2025-07-18T23:21:52Z","cross_cats_sorted":[],"title_canon_sha256":"6424f080eeda5d874ce14b4dc8539e2fb77468196fd7b90d7bd275da291e1001","abstract_canon_sha256":"c7ef282c7b463af5942576f8d8e2f0b34851d68147af4e8481c79db7f2e7b1d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:05.647803Z","signature_b64":"Q5UmflGLCdFuvF4JuvnL9b6vXPJhj1h8hQzWWbitqCqmZs5ab/g5p+6az5rIWRO4w/+6qwmTMmbiX9Mz870WDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eab5c7e78d1e2077a3445656def4770a0bd479363c10563cc795d9e908785a78","last_reissued_at":"2026-07-05T11:40:05.647276Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:05.647276Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NPUEval: Optimizing NPU Kernels with LLMs and Open Source Compilers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.PL","authors_text":"Graham Schelle, Sarunas Kalade","submitted_at":"2025-07-18T23:21:52Z","abstract_excerpt":"Neural processing units (NPUs) are gaining prominence in power-sensitive devices like client devices, with AI PCs being defined by their inclusion of these specialized processors. Running AI workloads efficiently on these devices requires libraries of optimized kernels. Creating efficient kernels demands expertise in domain-specific C++ with vector intrinsics and in-depth knowledge of the target architecture. Unlike GPU programming, which has had years to mature, NPU programming is new, with smaller and more fragmented developer communities across hardware platforms. This fragmentation poses a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.14403","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.14403/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.14403","created_at":"2026-07-05T11:40:05.647341+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.14403v1","created_at":"2026-07-05T11:40:05.647341+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.14403","created_at":"2026-07-05T11:40:05.647341+00:00"},{"alias_kind":"pith_short_12","alias_value":"5K24PZ4NDYQH","created_at":"2026-07-05T11:40:05.647341+00:00"},{"alias_kind":"pith_short_16","alias_value":"5K24PZ4NDYQHPI2E","created_at":"2026-07-05T11:40:05.647341+00:00"},{"alias_kind":"pith_short_8","alias_value":"5K24PZ4N","created_at":"2026-07-05T11:40:05.647341+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09708","citing_title":"Metal-Sci: A Scientific Compute Benchmark for Evolutionary LLM Kernel Search on Apple Silicon","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07586","citing_title":"From Human Guidance to Autonomy: Agent Skill System for End-to-End LLM Deployment on Spatial NPUs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23215","citing_title":"FastKernels: Benchmarking GPU Kernel Generation in Production","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5K24PZ4NDYQHPI2EKZLN55DXBI","json":"https://pith.science/pith/5K24PZ4NDYQHPI2EKZLN55DXBI.json","graph_json":"https://pith.science/api/pith-number/5K24PZ4NDYQHPI2EKZLN55DXBI/graph.json","events_json":"https://pith.science/api/pith-number/5K24PZ4NDYQHPI2EKZLN55DXBI/events.json","paper":"https://pith.science/paper/5K24PZ4N"},"agent_actions":{"view_html":"https://pith.science/pith/5K24PZ4NDYQHPI2EKZLN55DXBI","download_json":"https://pith.science/pith/5K24PZ4NDYQHPI2EKZLN55DXBI.json","view_paper":"https://pith.science/paper/5K24PZ4N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.14403&json=true","fetch_graph":"https://pith.science/api/pith-number/5K24PZ4NDYQHPI2EKZLN55DXBI/graph.json","fetch_events":"https://pith.science/api/pith-number/5K24PZ4NDYQHPI2EKZLN55DXBI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5K24PZ4NDYQHPI2EKZLN55DXBI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5K24PZ4NDYQHPI2EKZLN55DXBI/action/storage_attestation","attest_author":"https://pith.science/pith/5K24PZ4NDYQHPI2EKZLN55DXBI/action/author_attestation","sign_citation":"https://pith.science/pith/5K24PZ4NDYQHPI2EKZLN55DXBI/action/citation_signature","submit_replication":"https://pith.science/pith/5K24PZ4NDYQHPI2EKZLN55DXBI/action/replication_record"}},"created_at":"2026-07-05T11:40:05.647341+00:00","updated_at":"2026-07-05T11:40:05.647341+00:00"}