{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:RRNQQP3VOK7V2XZNRFM7UX3WPB","short_pith_number":"pith:RRNQQP3V","canonical_record":{"source":{"id":"2604.03750","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2026-04-04T14:51:09Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"098aa7373ee783094b9792d99cf78b4032afabe1b7e17a4915c8e48820e39167","abstract_canon_sha256":"0bc742a6bf17fc889335d11e9f402352ff3c9da240c8923914d03b777f266f87"},"schema_version":"1.0"},"canonical_sha256":"8c5b083f7572bf5d5f2d8959fa5f767856528e26d8bc7dc534e76dc8a3986efa","source":{"kind":"arxiv","id":"2604.03750","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.03750","created_at":"2026-08-07T00:45:49Z"},{"alias_kind":"arxiv_version","alias_value":"2604.03750v2","created_at":"2026-08-07T00:45:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.03750","created_at":"2026-08-07T00:45:49Z"},{"alias_kind":"pith_short_12","alias_value":"RRNQQP3VOK7V","created_at":"2026-08-07T00:45:49Z"},{"alias_kind":"pith_short_16","alias_value":"RRNQQP3VOK7V2XZN","created_at":"2026-08-07T00:45:49Z"},{"alias_kind":"pith_short_8","alias_value":"RRNQQP3V","created_at":"2026-08-07T00:45:49Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:RRNQQP3VOK7V2XZNRFM7UX3WPB","target":"record","payload":{"canonical_record":{"source":{"id":"2604.03750","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2026-04-04T14:51:09Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"098aa7373ee783094b9792d99cf78b4032afabe1b7e17a4915c8e48820e39167","abstract_canon_sha256":"0bc742a6bf17fc889335d11e9f402352ff3c9da240c8923914d03b777f266f87"},"schema_version":"1.0"},"canonical_sha256":"8c5b083f7572bf5d5f2d8959fa5f767856528e26d8bc7dc534e76dc8a3986efa","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-07T00:45:49.440394Z","signature_b64":"YFFwgrfcrUaNfWKA+DLO9S75FZpoCt/1Mit4x63KQRm1G5ZNPGfEPWe5EDhYk+DcBsRVQ4D7eWIMrWi7TvoMAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c5b083f7572bf5d5f2d8959fa5f767856528e26d8bc7dc534e76dc8a3986efa","last_reissued_at":"2026-08-07T00:45:49.438963Z","signature_status":"signed_v1","first_computed_at":"2026-08-07T00:45:49.438963Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.03750","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-07T00:45:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+cSoAyu/azAv8bx2+Vtv+OcWIOqjfotvaw7/FA1Y+RSkqSleFrEDHnALT1IqXnc4FulI2wcEE5JPtS+8e5G/AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T19:23:34.870589Z"},"content_sha256":"b3807203a19e438585cbecec372c2838bd1d0ac6823b2d8d7cc20223de81d880","schema_version":"1.0","event_id":"sha256:b3807203a19e438585cbecec372c2838bd1d0ac6823b2d8d7cc20223de81d880"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:RRNQQP3VOK7V2XZNRFM7UX3WPB","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"CREBench: Evaluating Large Language Models in Cryptographic Binary Reverse Engineering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Large language models achieve up to 64 points on a new benchmark for cryptographic binary reverse engineering, while human experts score 92.","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Baicheng Chen, Juanru Li, Tianxing He, Xiangru Liu, Yilei Chen, Yu Wang, Ziheng Zhou","submitted_at":"2026-04-04T14:51:09Z","abstract_excerpt":"Reverse engineering (RE) is central to software security, particularly for cryptographic programs that handle sensitive data and are highly prone to vulnerabilities. It supports critical tasks such as vulnerability discovery and malware analysis. Despite its importance, RE remains labor-intensive and requires substantial expertise, making large language models (LLMs) a potential solution for automating the process. However, their capabilities for RE remain systematically underexplored. To address this gap, we study the cryptographic binary RE capabilities of LLMs and introduce CREBench, a benc"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"GPT-5.4, the best-performing model, achieves 64.03 out of 100 and recovers the flag in 59% of challenges. We also establish a strong human expert baseline of 92.19 points.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The 432 challenges, built from 48 algorithms and three insecure key-usage scenarios, accurately represent the distribution and difficulty of real-world cryptographic binary reverse engineering without introducing unintended biases in the evaluation framework.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"CREBench benchmark finds frontier LLMs recover cryptographic flags in 59% of cases versus 92% for human experts.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Large language models achieve up to 64 points on a new benchmark for cryptographic binary reverse engineering, while human experts score 92.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"bb3a472cf39b975d5adc0aab207f43fa2c108edbcf9d91d7d4a1d38c42b3500a"},"source":{"id":"2604.03750","kind":"arxiv","version":2},"verdict":{"id":"99bf250d-7ab6-4767-9564-c205072d6704","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-13T17:25:01.090989Z","strongest_claim":"GPT-5.4, the best-performing model, achieves 64.03 out of 100 and recovers the flag in 59% of challenges. We also establish a strong human expert baseline of 92.19 points.","one_line_summary":"CREBench benchmark finds frontier LLMs recover cryptographic flags in 59% of cases versus 92% for human experts.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The 432 challenges, built from 48 algorithms and three insecure key-usage scenarios, accurately represent the distribution and difficulty of real-world cryptographic binary reverse engineering without introducing unintended biases in the evaluation framework.","pith_extraction_headline":"Large language models achieve up to 64 points on a new benchmark for cryptographic binary reverse engineering, while human experts score 92."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.03750/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"99bf250d-7ab6-4767-9564-c205072d6704"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-07T00:45:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"uXN/ajIU0ZInxCNRM6EiO/bg2ZpBquzJ6WGzfvepZOiQ6+9pEbDzTFDBwicgq6EFGplHP2IIWGXGKJ1v5tvKCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T19:23:34.871240Z"},"content_sha256":"580ecd98b48949963cbb9b56006398d0d02ad3f9913f396a71b074a667410a2c","schema_version":"1.0","event_id":"sha256:580ecd98b48949963cbb9b56006398d0d02ad3f9913f396a71b074a667410a2c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RRNQQP3VOK7V2XZNRFM7UX3WPB/bundle.json","state_url":"https://pith.science/pith/RRNQQP3VOK7V2XZNRFM7UX3WPB/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RRNQQP3VOK7V2XZNRFM7UX3WPB/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-20T19:23:34Z","links":{"resolver":"https://pith.science/pith/RRNQQP3VOK7V2XZNRFM7UX3WPB","bundle":"https://pith.science/pith/RRNQQP3VOK7V2XZNRFM7UX3WPB/bundle.json","state":"https://pith.science/pith/RRNQQP3VOK7V2XZNRFM7UX3WPB/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RRNQQP3VOK7V2XZNRFM7UX3WPB/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:RRNQQP3VOK7V2XZNRFM7UX3WPB","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0bc742a6bf17fc889335d11e9f402352ff3c9da240c8923914d03b777f266f87","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2026-04-04T14:51:09Z","title_canon_sha256":"098aa7373ee783094b9792d99cf78b4032afabe1b7e17a4915c8e48820e39167"},"schema_version":"1.0","source":{"id":"2604.03750","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.03750","created_at":"2026-08-07T00:45:49Z"},{"alias_kind":"arxiv_version","alias_value":"2604.03750v2","created_at":"2026-08-07T00:45:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.03750","created_at":"2026-08-07T00:45:49Z"},{"alias_kind":"pith_short_12","alias_value":"RRNQQP3VOK7V","created_at":"2026-08-07T00:45:49Z"},{"alias_kind":"pith_short_16","alias_value":"RRNQQP3VOK7V2XZN","created_at":"2026-08-07T00:45:49Z"},{"alias_kind":"pith_short_8","alias_value":"RRNQQP3V","created_at":"2026-08-07T00:45:49Z"}],"graph_snapshots":[{"event_id":"sha256:580ecd98b48949963cbb9b56006398d0d02ad3f9913f396a71b074a667410a2c","target":"graph","created_at":"2026-08-07T00:45:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"GPT-5.4, the best-performing model, achieves 64.03 out of 100 and recovers the flag in 59% of challenges. We also establish a strong human expert baseline of 92.19 points."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The 432 challenges, built from 48 algorithms and three insecure key-usage scenarios, accurately represent the distribution and difficulty of real-world cryptographic binary reverse engineering without introducing unintended biases in the evaluation framework."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"CREBench benchmark finds frontier LLMs recover cryptographic flags in 59% of cases versus 92% for human experts."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Large language models achieve up to 64 points on a new benchmark for cryptographic binary reverse engineering, while human experts score 92."}],"snapshot_sha256":"bb3a472cf39b975d5adc0aab207f43fa2c108edbcf9d91d7d4a1d38c42b3500a"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2604.03750/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reverse engineering (RE) is central to software security, particularly for cryptographic programs that handle sensitive data and are highly prone to vulnerabilities. It supports critical tasks such as vulnerability discovery and malware analysis. Despite its importance, RE remains labor-intensive and requires substantial expertise, making large language models (LLMs) a potential solution for automating the process. However, their capabilities for RE remain systematically underexplored. To address this gap, we study the cryptographic binary RE capabilities of LLMs and introduce CREBench, a benc","authors_text":"Baicheng Chen, Juanru Li, Tianxing He, Xiangru Liu, Yilei Chen, Yu Wang, Ziheng Zhou","cross_cats":["cs.AI","cs.CL"],"headline":"Large language models achieve up to 64 points on a new benchmark for cryptographic binary reverse engineering, while human experts score 92.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2026-04-04T14:51:09Z","title":"CREBench: Evaluating Large Language Models in Cryptographic Binary Reverse Engineering"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.03750","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-13T17:25:01.090989Z","id":"99bf250d-7ab6-4767-9564-c205072d6704","model_set":{"reader":"grok-4.3"},"one_line_summary":"CREBench benchmark finds frontier LLMs recover cryptographic flags in 59% of cases versus 92% for human experts.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Large language models achieve up to 64 points on a new benchmark for cryptographic binary reverse engineering, while human experts score 92.","strongest_claim":"GPT-5.4, the best-performing model, achieves 64.03 out of 100 and recovers the flag in 59% of challenges. We also establish a strong human expert baseline of 92.19 points.","weakest_assumption":"The 432 challenges, built from 48 algorithms and three insecure key-usage scenarios, accurately represent the distribution and difficulty of real-world cryptographic binary reverse engineering without introducing unintended biases in the evaluation framework."}},"verdict_id":"99bf250d-7ab6-4767-9564-c205072d6704"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b3807203a19e438585cbecec372c2838bd1d0ac6823b2d8d7cc20223de81d880","target":"record","created_at":"2026-08-07T00:45:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0bc742a6bf17fc889335d11e9f402352ff3c9da240c8923914d03b777f266f87","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2026-04-04T14:51:09Z","title_canon_sha256":"098aa7373ee783094b9792d99cf78b4032afabe1b7e17a4915c8e48820e39167"},"schema_version":"1.0","source":{"id":"2604.03750","kind":"arxiv","version":2}},"canonical_sha256":"8c5b083f7572bf5d5f2d8959fa5f767856528e26d8bc7dc534e76dc8a3986efa","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"8c5b083f7572bf5d5f2d8959fa5f767856528e26d8bc7dc534e76dc8a3986efa","first_computed_at":"2026-08-07T00:45:49.438963Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-08-07T00:45:49.438963Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"YFFwgrfcrUaNfWKA+DLO9S75FZpoCt/1Mit4x63KQRm1G5ZNPGfEPWe5EDhYk+DcBsRVQ4D7eWIMrWi7TvoMAQ==","signature_status":"signed_v1","signed_at":"2026-08-07T00:45:49.440394Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.03750","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b3807203a19e438585cbecec372c2838bd1d0ac6823b2d8d7cc20223de81d880","sha256:580ecd98b48949963cbb9b56006398d0d02ad3f9913f396a71b074a667410a2c"],"state_sha256":"e8332093cf8777b79e3b14b9bc77d1303138252d6a4f28178de9288d756d3984"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xUPGyD1JmqtSYB1+B0By8iFtH46QCqnrEmKr7PbX1nA04GNwwyZ/2ctYJ9pfG8kYgBFuJu9J1OIIoyBaCxktAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-20T19:23:34.875591Z","bundle_sha256":"983882740c5a3318a5f6f109d2e058dd3e7a5cb6227621fa488f33ae17f100ec"}}