{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:5U7KA5QGW7RJLT33OWQXOAS675","short_pith_number":"pith:5U7KA5QG","canonical_record":{"source":{"id":"2310.16049","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-24T17:59:20Z","cross_cats_sorted":[],"title_canon_sha256":"e1d2c19f73f22c8c1563d4592d87d0a40fbb8593ddbc3114359fbf134afdd28f","abstract_canon_sha256":"df5798cd73c34fe3e1375e74de013739f485f0da4d9662222e0fed39a4b70d3f"},"schema_version":"1.0"},"canonical_sha256":"ed3ea07606b7e295cf7b75a177025eff409cdac147bc7e51a283619c1c9669a2","source":{"kind":"arxiv","id":"2310.16049","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.16049","created_at":"2026-07-05T07:59:55Z"},{"alias_kind":"arxiv_version","alias_value":"2310.16049v2","created_at":"2026-07-05T07:59:55Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.16049","created_at":"2026-07-05T07:59:55Z"},{"alias_kind":"pith_short_12","alias_value":"5U7KA5QGW7RJ","created_at":"2026-07-05T07:59:55Z"},{"alias_kind":"pith_short_16","alias_value":"5U7KA5QGW7RJLT33","created_at":"2026-07-05T07:59:55Z"},{"alias_kind":"pith_short_8","alias_value":"5U7KA5QG","created_at":"2026-07-05T07:59:55Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:5U7KA5QGW7RJLT33OWQXOAS675","target":"record","payload":{"canonical_record":{"source":{"id":"2310.16049","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-24T17:59:20Z","cross_cats_sorted":[],"title_canon_sha256":"e1d2c19f73f22c8c1563d4592d87d0a40fbb8593ddbc3114359fbf134afdd28f","abstract_canon_sha256":"df5798cd73c34fe3e1375e74de013739f485f0da4d9662222e0fed39a4b70d3f"},"schema_version":"1.0"},"canonical_sha256":"ed3ea07606b7e295cf7b75a177025eff409cdac147bc7e51a283619c1c9669a2","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:59:55.505170Z","signature_b64":"UwlMWkdE0JupBSZ5ylgtfmPjmaGW03vuL3yNYpmIGHgMeqnZ/HKz0YzPenl0UTQBoBujgfSmhifNmSgy5K8rDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed3ea07606b7e295cf7b75a177025eff409cdac147bc7e51a283619c1c9669a2","last_reissued_at":"2026-07-05T07:59:55.504641Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:59:55.504641Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2310.16049","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:59:55Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Nk2qt4kHKnJINcf6B4yeheQwsEc9gZUS5WOF2jzbQALOL+OHEjUu2Ifh/AosvVNKWonx4/4AOG4hwC3bHy17Bg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T12:39:03.675001Z"},"content_sha256":"dcf2ae5650380288a6b251095ededba2b7ea4dc57593144b5f7e4370e227af30","schema_version":"1.0","event_id":"sha256:dcf2ae5650380288a6b251095ededba2b7ea4dc57593144b5f7e4370e227af30"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:5U7KA5QGW7RJLT33OWQXOAS675","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"MuSR: Testing the Limits of Chain-of-thought with Multistep Soft Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Greg Durrett, Kaj Bostrom, Swarat Chaudhuri, Xi Ye, Zayne Sprague","submitted_at":"2023-10-24T17:59:20Z","abstract_excerpt":"While large language models (LLMs) equipped with techniques like chain-of-thought prompting have demonstrated impressive capabilities, they still fall short in their ability to reason robustly in complex settings. However, evaluating LLM reasoning is challenging because system capabilities continue to grow while benchmark datasets for tasks like logical deduction have remained static. We introduce MuSR, a dataset for evaluating language models on multistep soft reasoning tasks specified in a natural language narrative. This dataset has two crucial features. First, it is created through a novel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.16049","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.16049/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:59:55Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zR3PlCXDvOOIwIy6nKqEShdvv3H6EWrtDQKJAdKFUS+8CGFoQy9z2T96ICRiUbTIpuNgPWeus/wgBlPGmDuEDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T12:39:03.675478Z"},"content_sha256":"638b95841a62dcc39920403876e7bda419ef3392e843d7e39ab815b29dda63bc","schema_version":"1.0","event_id":"sha256:638b95841a62dcc39920403876e7bda419ef3392e843d7e39ab815b29dda63bc"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/5U7KA5QGW7RJLT33OWQXOAS675/bundle.json","state_url":"https://pith.science/pith/5U7KA5QGW7RJLT33OWQXOAS675/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/5U7KA5QGW7RJLT33OWQXOAS675/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T12:39:03Z","links":{"resolver":"https://pith.science/pith/5U7KA5QGW7RJLT33OWQXOAS675","bundle":"https://pith.science/pith/5U7KA5QGW7RJLT33OWQXOAS675/bundle.json","state":"https://pith.science/pith/5U7KA5QGW7RJLT33OWQXOAS675/state.json","well_known_bundle":"https://pith.science/.well-known/pith/5U7KA5QGW7RJLT33OWQXOAS675/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:5U7KA5QGW7RJLT33OWQXOAS675","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"df5798cd73c34fe3e1375e74de013739f485f0da4d9662222e0fed39a4b70d3f","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-24T17:59:20Z","title_canon_sha256":"e1d2c19f73f22c8c1563d4592d87d0a40fbb8593ddbc3114359fbf134afdd28f"},"schema_version":"1.0","source":{"id":"2310.16049","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.16049","created_at":"2026-07-05T07:59:55Z"},{"alias_kind":"arxiv_version","alias_value":"2310.16049v2","created_at":"2026-07-05T07:59:55Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.16049","created_at":"2026-07-05T07:59:55Z"},{"alias_kind":"pith_short_12","alias_value":"5U7KA5QGW7RJ","created_at":"2026-07-05T07:59:55Z"},{"alias_kind":"pith_short_16","alias_value":"5U7KA5QGW7RJLT33","created_at":"2026-07-05T07:59:55Z"},{"alias_kind":"pith_short_8","alias_value":"5U7KA5QG","created_at":"2026-07-05T07:59:55Z"}],"graph_snapshots":[{"event_id":"sha256:638b95841a62dcc39920403876e7bda419ef3392e843d7e39ab815b29dda63bc","target":"graph","created_at":"2026-07-05T07:59:55Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2310.16049/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"While large language models (LLMs) equipped with techniques like chain-of-thought prompting have demonstrated impressive capabilities, they still fall short in their ability to reason robustly in complex settings. However, evaluating LLM reasoning is challenging because system capabilities continue to grow while benchmark datasets for tasks like logical deduction have remained static. We introduce MuSR, a dataset for evaluating language models on multistep soft reasoning tasks specified in a natural language narrative. This dataset has two crucial features. First, it is created through a novel","authors_text":"Greg Durrett, Kaj Bostrom, Swarat Chaudhuri, Xi Ye, Zayne Sprague","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-24T17:59:20Z","title":"MuSR: Testing the Limits of Chain-of-thought with Multistep Soft Reasoning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.16049","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:dcf2ae5650380288a6b251095ededba2b7ea4dc57593144b5f7e4370e227af30","target":"record","created_at":"2026-07-05T07:59:55Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"df5798cd73c34fe3e1375e74de013739f485f0da4d9662222e0fed39a4b70d3f","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-24T17:59:20Z","title_canon_sha256":"e1d2c19f73f22c8c1563d4592d87d0a40fbb8593ddbc3114359fbf134afdd28f"},"schema_version":"1.0","source":{"id":"2310.16049","kind":"arxiv","version":2}},"canonical_sha256":"ed3ea07606b7e295cf7b75a177025eff409cdac147bc7e51a283619c1c9669a2","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ed3ea07606b7e295cf7b75a177025eff409cdac147bc7e51a283619c1c9669a2","first_computed_at":"2026-07-05T07:59:55.504641Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:59:55.504641Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"UwlMWkdE0JupBSZ5ylgtfmPjmaGW03vuL3yNYpmIGHgMeqnZ/HKz0YzPenl0UTQBoBujgfSmhifNmSgy5K8rDw==","signature_status":"signed_v1","signed_at":"2026-07-05T07:59:55.505170Z","signed_message":"canonical_sha256_bytes"},"source_id":"2310.16049","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:dcf2ae5650380288a6b251095ededba2b7ea4dc57593144b5f7e4370e227af30","sha256:638b95841a62dcc39920403876e7bda419ef3392e843d7e39ab815b29dda63bc"],"state_sha256":"0aad958574b6da64099e19257637b3c4660a5e775d4d6f4ba8dff59002bb9b96"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"87uK3oYnNpdswS1EPig8QwIgW5Bibi2XssgwMtDrWWFvza9FnhhDscD3LFVZLYWmQ/p0CxPZ//SMMExuPqvFBQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T12:39:03.679158Z","bundle_sha256":"2e7bfcfaa1c38c41948749dc79ddd013a309cd00d67e63bd38341aa688f0cfe2"}}