{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BA4DCPY7IBS6WGZGXIV3W25BTE","short_pith_number":"pith:BA4DCPY7","schema_version":"1.0","canonical_sha256":"0838313f1f4065eb1b26ba2bbb6ba1990389ab2dd34136eff1ba116c0bb2470e","source":{"kind":"arxiv","id":"2503.15242","version":2},"attestation_state":"computed","paper":{"title":"BigO(Bench) -- Can LLMs Generate Code with Controlled Time and Space Complexity?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CC"],"primary_cat":"cs.CL","authors_text":"Baptiste Roziere, Benoit Sagot, Gabriel Synnaeve, Pierre Chambon","submitted_at":"2025-03-19T14:19:57Z","abstract_excerpt":"We introduce BigO(Bench), a novel coding benchmark designed to evaluate the capabilities of generative language models in understanding and generating code with specified time and space complexities. This benchmark addresses the gap in current evaluations that often overlook the ability of models to comprehend and produce code constrained by computational complexity. BigO(Bench) includes tooling to infer the algorithmic complexity of any Python function from profiling measurements, including human- or LLM-generated solutions. BigO(Bench) also includes of set of 3,105 coding problems and 1,190,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.15242","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-19T14:19:57Z","cross_cats_sorted":["cs.AI","cs.CC"],"title_canon_sha256":"3382c15eaf04e486a4854cce6ffe45c8e57eedbaccd24573aad77df3c4648f99","abstract_canon_sha256":"6594d1172c9f6d08715dd61bfa50d24f8cdff25330e3a8b7f7bd0737724ab0ef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:36:16.260087Z","signature_b64":"vW4zaVLuio3rMeOw0JhZIcMlqJhjHln8ESXeNPSYRtfFfbnh4Ih/h8mhJ4hywmUgR+AMo4Go2F7j11PwV7L3AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0838313f1f4065eb1b26ba2bbb6ba1990389ab2dd34136eff1ba116c0bb2470e","last_reissued_at":"2026-07-05T10:36:16.259010Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:36:16.259010Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BigO(Bench) -- Can LLMs Generate Code with Controlled Time and Space Complexity?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CC"],"primary_cat":"cs.CL","authors_text":"Baptiste Roziere, Benoit Sagot, Gabriel Synnaeve, Pierre Chambon","submitted_at":"2025-03-19T14:19:57Z","abstract_excerpt":"We introduce BigO(Bench), a novel coding benchmark designed to evaluate the capabilities of generative language models in understanding and generating code with specified time and space complexities. This benchmark addresses the gap in current evaluations that often overlook the ability of models to comprehend and produce code constrained by computational complexity. BigO(Bench) includes tooling to infer the algorithmic complexity of any Python function from profiling measurements, including human- or LLM-generated solutions. BigO(Bench) also includes of set of 3,105 coding problems and 1,190,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.15242","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.15242/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.15242","created_at":"2026-07-05T10:36:16.259155+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.15242v2","created_at":"2026-07-05T10:36:16.259155+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.15242","created_at":"2026-07-05T10:36:16.259155+00:00"},{"alias_kind":"pith_short_12","alias_value":"BA4DCPY7IBS6","created_at":"2026-07-05T10:36:16.259155+00:00"},{"alias_kind":"pith_short_16","alias_value":"BA4DCPY7IBS6WGZG","created_at":"2026-07-05T10:36:16.259155+00:00"},{"alias_kind":"pith_short_8","alias_value":"BA4DCPY7","created_at":"2026-07-05T10:36:16.259155+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06826","citing_title":"SkelDPO: A Skeleton-Guided Direct Preference Optimization Framework for Efficient Code Generation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06821","citing_title":"Chiseling Out Efficiency: Structured Skeleton Supervision for Efficient Code Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28751","citing_title":"Extrapolative Weight Averaging Reveals Correctness-Efficiency Frontiers in Code RL","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30394","citing_title":"CodeGolf Bench: A Multi-Language Benchmark for Evaluating Concise Code Generation Capabilities of Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.11687","citing_title":"MetaLint: Easy-to-Hard Generalization for Code Linting","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BA4DCPY7IBS6WGZGXIV3W25BTE","json":"https://pith.science/pith/BA4DCPY7IBS6WGZGXIV3W25BTE.json","graph_json":"https://pith.science/api/pith-number/BA4DCPY7IBS6WGZGXIV3W25BTE/graph.json","events_json":"https://pith.science/api/pith-number/BA4DCPY7IBS6WGZGXIV3W25BTE/events.json","paper":"https://pith.science/paper/BA4DCPY7"},"agent_actions":{"view_html":"https://pith.science/pith/BA4DCPY7IBS6WGZGXIV3W25BTE","download_json":"https://pith.science/pith/BA4DCPY7IBS6WGZGXIV3W25BTE.json","view_paper":"https://pith.science/paper/BA4DCPY7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.15242&json=true","fetch_graph":"https://pith.science/api/pith-number/BA4DCPY7IBS6WGZGXIV3W25BTE/graph.json","fetch_events":"https://pith.science/api/pith-number/BA4DCPY7IBS6WGZGXIV3W25BTE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BA4DCPY7IBS6WGZGXIV3W25BTE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BA4DCPY7IBS6WGZGXIV3W25BTE/action/storage_attestation","attest_author":"https://pith.science/pith/BA4DCPY7IBS6WGZGXIV3W25BTE/action/author_attestation","sign_citation":"https://pith.science/pith/BA4DCPY7IBS6WGZGXIV3W25BTE/action/citation_signature","submit_replication":"https://pith.science/pith/BA4DCPY7IBS6WGZGXIV3W25BTE/action/replication_record"}},"created_at":"2026-07-05T10:36:16.259155+00:00","updated_at":"2026-07-05T10:36:16.259155+00:00"}