{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VAU65ZZJVFPDTCL3D5TVLWZVB7","short_pith_number":"pith:VAU65ZZJ","schema_version":"1.0","canonical_sha256":"a829eee729a95e39897b1f6755db350fe642f65a603b3421d59aa275fe8e2cb0","source":{"kind":"arxiv","id":"2504.20183","version":1},"attestation_state":"computed","paper":{"title":"BLADE: Benchmark suite for LLM-driven Automated Design and Evolution of iterative optimisation heuristics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.NE"],"primary_cat":"cs.SE","authors_text":"Anna V. Kononova, Haoran Yin, Niki van Stein, Thomas B\\\"ack","submitted_at":"2025-04-28T18:34:09Z","abstract_excerpt":"The application of Large Language Models (LLMs) for Automated Algorithm Discovery (AAD), particularly for optimisation heuristics, is an emerging field of research. This emergence necessitates robust, standardised benchmarking practices to rigorously evaluate the capabilities and limitations of LLM-driven AAD methods and the resulting generated algorithms, especially given the opacity of their design process and known issues with existing benchmarks. To address this need, we introduce BLADE (Benchmark suite for LLM-driven Automated Design and Evolution), a modular and extensible framework spec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.20183","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-04-28T18:34:09Z","cross_cats_sorted":["cs.AI","cs.NE"],"title_canon_sha256":"de7612b94d096840901eef895623fdc47c16cd288aba5ae56ce550052adc05d9","abstract_canon_sha256":"4a55e679223e833563b764451a0695c1dbf60011303ec4d9f4d71ef9097ce7f3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:55:37.989138Z","signature_b64":"SH5QW3TwLMVa5qT/KFhB1QXYm0/QgSkuXRUJzpLdflVBoK9frMtdenZGAa+QoaYR6ktlwKWRSDeI0KZL4N0qAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a829eee729a95e39897b1f6755db350fe642f65a603b3421d59aa275fe8e2cb0","last_reissued_at":"2026-07-05T10:55:37.988674Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:55:37.988674Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BLADE: Benchmark suite for LLM-driven Automated Design and Evolution of iterative optimisation heuristics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.NE"],"primary_cat":"cs.SE","authors_text":"Anna V. Kononova, Haoran Yin, Niki van Stein, Thomas B\\\"ack","submitted_at":"2025-04-28T18:34:09Z","abstract_excerpt":"The application of Large Language Models (LLMs) for Automated Algorithm Discovery (AAD), particularly for optimisation heuristics, is an emerging field of research. This emergence necessitates robust, standardised benchmarking practices to rigorously evaluate the capabilities and limitations of LLM-driven AAD methods and the resulting generated algorithms, especially given the opacity of their design process and known issues with existing benchmarks. To address this need, we introduce BLADE (Benchmark suite for LLM-driven Automated Design and Evolution), a modular and extensible framework spec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.20183","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.20183/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.20183","created_at":"2026-07-05T10:55:37.988725+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.20183v1","created_at":"2026-07-05T10:55:37.988725+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.20183","created_at":"2026-07-05T10:55:37.988725+00:00"},{"alias_kind":"pith_short_12","alias_value":"VAU65ZZJVFPD","created_at":"2026-07-05T10:55:37.988725+00:00"},{"alias_kind":"pith_short_16","alias_value":"VAU65ZZJVFPDTCL3","created_at":"2026-07-05T10:55:37.988725+00:00"},{"alias_kind":"pith_short_8","alias_value":"VAU65ZZJ","created_at":"2026-07-05T10:55:37.988725+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.03605","citing_title":"Behaviour Space Analysis of LLM-driven Meta-heuristic Discovery","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VAU65ZZJVFPDTCL3D5TVLWZVB7","json":"https://pith.science/pith/VAU65ZZJVFPDTCL3D5TVLWZVB7.json","graph_json":"https://pith.science/api/pith-number/VAU65ZZJVFPDTCL3D5TVLWZVB7/graph.json","events_json":"https://pith.science/api/pith-number/VAU65ZZJVFPDTCL3D5TVLWZVB7/events.json","paper":"https://pith.science/paper/VAU65ZZJ"},"agent_actions":{"view_html":"https://pith.science/pith/VAU65ZZJVFPDTCL3D5TVLWZVB7","download_json":"https://pith.science/pith/VAU65ZZJVFPDTCL3D5TVLWZVB7.json","view_paper":"https://pith.science/paper/VAU65ZZJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.20183&json=true","fetch_graph":"https://pith.science/api/pith-number/VAU65ZZJVFPDTCL3D5TVLWZVB7/graph.json","fetch_events":"https://pith.science/api/pith-number/VAU65ZZJVFPDTCL3D5TVLWZVB7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VAU65ZZJVFPDTCL3D5TVLWZVB7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VAU65ZZJVFPDTCL3D5TVLWZVB7/action/storage_attestation","attest_author":"https://pith.science/pith/VAU65ZZJVFPDTCL3D5TVLWZVB7/action/author_attestation","sign_citation":"https://pith.science/pith/VAU65ZZJVFPDTCL3D5TVLWZVB7/action/citation_signature","submit_replication":"https://pith.science/pith/VAU65ZZJVFPDTCL3D5TVLWZVB7/action/replication_record"}},"created_at":"2026-07-05T10:55:37.988725+00:00","updated_at":"2026-07-05T10:55:37.988725+00:00"}