{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:Z2EFW4S4NVZZAEEFAHI23HREE7","short_pith_number":"pith:Z2EFW4S4","canonical_record":{"source":{"id":"2410.17558","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-10-23T04:55:08Z","cross_cats_sorted":[],"title_canon_sha256":"917910957f93d165fd7304279b6a39f12dac454e91eeeb50a96a15c654413575","abstract_canon_sha256":"6f1685469d99f4682addbbd465f4c0bf3ebeb256b0af27063f3b909f9bdf71d1"},"schema_version":"1.0"},"canonical_sha256":"ce885b725c6d7390108501d1ad9e2427cce2ff402699dc65ba6deb5da46162e3","source":{"kind":"arxiv","id":"2410.17558","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.17558","created_at":"2026-07-05T09:25:48Z"},{"alias_kind":"arxiv_version","alias_value":"2410.17558v2","created_at":"2026-07-05T09:25:48Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.17558","created_at":"2026-07-05T09:25:48Z"},{"alias_kind":"pith_short_12","alias_value":"Z2EFW4S4NVZZ","created_at":"2026-07-05T09:25:48Z"},{"alias_kind":"pith_short_16","alias_value":"Z2EFW4S4NVZZAEEF","created_at":"2026-07-05T09:25:48Z"},{"alias_kind":"pith_short_8","alias_value":"Z2EFW4S4","created_at":"2026-07-05T09:25:48Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:Z2EFW4S4NVZZAEEFAHI23HREE7","target":"record","payload":{"canonical_record":{"source":{"id":"2410.17558","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-10-23T04:55:08Z","cross_cats_sorted":[],"title_canon_sha256":"917910957f93d165fd7304279b6a39f12dac454e91eeeb50a96a15c654413575","abstract_canon_sha256":"6f1685469d99f4682addbbd465f4c0bf3ebeb256b0af27063f3b909f9bdf71d1"},"schema_version":"1.0"},"canonical_sha256":"ce885b725c6d7390108501d1ad9e2427cce2ff402699dc65ba6deb5da46162e3","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:25:48.735180Z","signature_b64":"k+bCQqEkP8RhOs4vkFesb0AChdrSDec7cQiO6eDzEO8Pg6KSXYgoK2OTeIO8fhYP4M2Xx6B/UDUQFhxtMGPTDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce885b725c6d7390108501d1ad9e2427cce2ff402699dc65ba6deb5da46162e3","last_reissued_at":"2026-07-05T09:25:48.734702Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:25:48.734702Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2410.17558","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:25:48Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+X5FbP60/6EuP2cfzOgvQw7gWhaWvNN55E0t94BtzREinmliB/vJTm9CkP/cby3BONWa0v+IDBmxm2KS2Pe1Cg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T19:10:26.965867Z"},"content_sha256":"0375ffb21c7030cedc43a415d8896d1aa51223867716d9db871fcd7eee06ec04","schema_version":"1.0","event_id":"sha256:0375ffb21c7030cedc43a415d8896d1aa51223867716d9db871fcd7eee06ec04"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:Z2EFW4S4NVZZAEEFAHI23HREE7","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"CLR-Bench: Evaluating Large Language Models in College-level Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Feiran Huang, Junnan Dong, Xiao Huang, Xinrun Wang, Yuanchen Bei, Zijin Hong","submitted_at":"2024-10-23T04:55:08Z","abstract_excerpt":"Large language models (LLMs) have demonstrated their remarkable performance across various language understanding tasks. While emerging benchmarks have been proposed to evaluate LLMs in various domains such as mathematics and computer science, they merely measure the accuracy in terms of the final prediction on multi-choice questions. However, it remains insufficient to verify the essential understanding of LLMs given a chosen choice. To fill this gap, we present CLR-Bench to comprehensively evaluate the LLMs in complex college-level reasoning. Specifically, (i) we prioritize 16 challenging co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.17558","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.17558/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:25:48Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"FCvMGnUB7Y1by4mYYE6+rs7hACjC4CuOMQndS41HTOpbSgcMej438RLyKBWbPmNGL3lNObDBE0TnsXudXYiVDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T19:10:26.966832Z"},"content_sha256":"078f99a280f8b428c5bfad62d1a1ce29651db1a4d107d761c9f45b2c79d040f3","schema_version":"1.0","event_id":"sha256:078f99a280f8b428c5bfad62d1a1ce29651db1a4d107d761c9f45b2c79d040f3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/Z2EFW4S4NVZZAEEFAHI23HREE7/bundle.json","state_url":"https://pith.science/pith/Z2EFW4S4NVZZAEEFAHI23HREE7/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/Z2EFW4S4NVZZAEEFAHI23HREE7/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-19T19:10:26Z","links":{"resolver":"https://pith.science/pith/Z2EFW4S4NVZZAEEFAHI23HREE7","bundle":"https://pith.science/pith/Z2EFW4S4NVZZAEEFAHI23HREE7/bundle.json","state":"https://pith.science/pith/Z2EFW4S4NVZZAEEFAHI23HREE7/state.json","well_known_bundle":"https://pith.science/.well-known/pith/Z2EFW4S4NVZZAEEFAHI23HREE7/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:Z2EFW4S4NVZZAEEFAHI23HREE7","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6f1685469d99f4682addbbd465f4c0bf3ebeb256b0af27063f3b909f9bdf71d1","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-10-23T04:55:08Z","title_canon_sha256":"917910957f93d165fd7304279b6a39f12dac454e91eeeb50a96a15c654413575"},"schema_version":"1.0","source":{"id":"2410.17558","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.17558","created_at":"2026-07-05T09:25:48Z"},{"alias_kind":"arxiv_version","alias_value":"2410.17558v2","created_at":"2026-07-05T09:25:48Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.17558","created_at":"2026-07-05T09:25:48Z"},{"alias_kind":"pith_short_12","alias_value":"Z2EFW4S4NVZZ","created_at":"2026-07-05T09:25:48Z"},{"alias_kind":"pith_short_16","alias_value":"Z2EFW4S4NVZZAEEF","created_at":"2026-07-05T09:25:48Z"},{"alias_kind":"pith_short_8","alias_value":"Z2EFW4S4","created_at":"2026-07-05T09:25:48Z"}],"graph_snapshots":[{"event_id":"sha256:078f99a280f8b428c5bfad62d1a1ce29651db1a4d107d761c9f45b2c79d040f3","target":"graph","created_at":"2026-07-05T09:25:48Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.17558/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large language models (LLMs) have demonstrated their remarkable performance across various language understanding tasks. While emerging benchmarks have been proposed to evaluate LLMs in various domains such as mathematics and computer science, they merely measure the accuracy in terms of the final prediction on multi-choice questions. However, it remains insufficient to verify the essential understanding of LLMs given a chosen choice. To fill this gap, we present CLR-Bench to comprehensively evaluate the LLMs in complex college-level reasoning. Specifically, (i) we prioritize 16 challenging co","authors_text":"Feiran Huang, Junnan Dong, Xiao Huang, Xinrun Wang, Yuanchen Bei, Zijin Hong","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-10-23T04:55:08Z","title":"CLR-Bench: Evaluating Large Language Models in College-level Reasoning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.17558","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:0375ffb21c7030cedc43a415d8896d1aa51223867716d9db871fcd7eee06ec04","target":"record","created_at":"2026-07-05T09:25:48Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6f1685469d99f4682addbbd465f4c0bf3ebeb256b0af27063f3b909f9bdf71d1","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-10-23T04:55:08Z","title_canon_sha256":"917910957f93d165fd7304279b6a39f12dac454e91eeeb50a96a15c654413575"},"schema_version":"1.0","source":{"id":"2410.17558","kind":"arxiv","version":2}},"canonical_sha256":"ce885b725c6d7390108501d1ad9e2427cce2ff402699dc65ba6deb5da46162e3","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ce885b725c6d7390108501d1ad9e2427cce2ff402699dc65ba6deb5da46162e3","first_computed_at":"2026-07-05T09:25:48.734702Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:25:48.734702Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"k+bCQqEkP8RhOs4vkFesb0AChdrSDec7cQiO6eDzEO8Pg6KSXYgoK2OTeIO8fhYP4M2Xx6B/UDUQFhxtMGPTDg==","signature_status":"signed_v1","signed_at":"2026-07-05T09:25:48.735180Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.17558","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:0375ffb21c7030cedc43a415d8896d1aa51223867716d9db871fcd7eee06ec04","sha256:078f99a280f8b428c5bfad62d1a1ce29651db1a4d107d761c9f45b2c79d040f3"],"state_sha256":"c5506e71ba39cf244a12fbe8d77be2ddd7ff5edbeae6dfee72ded4ae6ffae8b2"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"KOipt5cIUMNXU2ORkTSvqgaBjkwMOMUg/ZdgNAX4Zu7pACS7l7yGnGnl8h75ihpkYw+PdZ04pSclPuqPl5TUCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-19T19:10:26.975733Z","bundle_sha256":"e4fee742f7b1950b8cfebe7f34a70d26ed908355357acd74f48b5d5b8765f8c1"}}