{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:IXC5F7PDSFYDQVS7IOM4RZQ44W","short_pith_number":"pith:IXC5F7PD","schema_version":"1.0","canonical_sha256":"45c5d2fde3917038565f4399c8e61ce5a712815543618bf19c58ff55ebc23e76","source":{"kind":"arxiv","id":"2212.07919","version":2},"attestation_state":"computed","paper":{"title":"ROSCOE: A Suite of Metrics for Scoring Step-by-Step Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Asli Celikyilmaz, Luke Zettlemoyer, Martin Corredor, Maryam Fazel-Zarandi, Moya Chen, Olga Golovneva, Spencer Poff","submitted_at":"2022-12-15T15:52:39Z","abstract_excerpt":"Large language models show improved downstream task performance when prompted to generate step-by-step reasoning to justify their final answers. These reasoning steps greatly improve model interpretability and verification, but objectively studying their correctness (independent of the final answer) is difficult without reliable methods for automatic evaluation. We simply do not know how often the stated reasoning steps actually support the final end task predictions. In this work, we present ROSCOE, a suite of interpretable, unsupervised automatic scores that improve and extend previous text "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.07919","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-12-15T15:52:39Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"06f60e21f4c5e28452ac47ba5048b0184059ccc9bd4edb974b7fefc41ceff51e","abstract_canon_sha256":"fb9af92dfca3e190b14c7d16e3b1baa668cd0a65acac7d79474cfa6ea5f82fdd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:49:58.595849Z","signature_b64":"nPC9fQWSvCH9CFBk0abk3knmyZpTkYkK6G7V7TtdZWGC2JxQV6Fh6UJeDEdO7m6OFLAiWdpfuQt6KkBhJfnIAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45c5d2fde3917038565f4399c8e61ce5a712815543618bf19c58ff55ebc23e76","last_reissued_at":"2026-07-05T06:49:58.595222Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:49:58.595222Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ROSCOE: A Suite of Metrics for Scoring Step-by-Step Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Asli Celikyilmaz, Luke Zettlemoyer, Martin Corredor, Maryam Fazel-Zarandi, Moya Chen, Olga Golovneva, Spencer Poff","submitted_at":"2022-12-15T15:52:39Z","abstract_excerpt":"Large language models show improved downstream task performance when prompted to generate step-by-step reasoning to justify their final answers. These reasoning steps greatly improve model interpretability and verification, but objectively studying their correctness (independent of the final answer) is difficult without reliable methods for automatic evaluation. We simply do not know how often the stated reasoning steps actually support the final end task predictions. In this work, we present ROSCOE, a suite of interpretable, unsupervised automatic scores that improve and extend previous text "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.07919","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.07919/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.07919","created_at":"2026-07-05T06:49:58.595290+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.07919v2","created_at":"2026-07-05T06:49:58.595290+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.07919","created_at":"2026-07-05T06:49:58.595290+00:00"},{"alias_kind":"pith_short_12","alias_value":"IXC5F7PDSFYD","created_at":"2026-07-05T06:49:58.595290+00:00"},{"alias_kind":"pith_short_16","alias_value":"IXC5F7PDSFYDQVS7","created_at":"2026-07-05T06:49:58.595290+00:00"},{"alias_kind":"pith_short_8","alias_value":"IXC5F7PD","created_at":"2026-07-05T06:49:58.595290+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10279","citing_title":"Supervised Fine-tuning with Synthetic Rationale Data Hurts Real-World Disease Prediction","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31608","citing_title":"CLExEval: A Human-in-the-Loop Framework for Qualitative Evaluation of LLM Clinical Reasoning","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29278","citing_title":"The Complexity Ceiling Benchmark: A Multi-Domain Evaluation of Sequential Reasoning Under Depth Scaling","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2509.25844","citing_title":"Believing without Seeing: Quality Scores for Contextualizing Vision-Language Model Explanations","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03332","citing_title":"Fragile Thoughts: How Large Language Models Handle Chain-of-Thought Perturbations","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06066","citing_title":"From Hallucination to Structure Snowballing: The Alignment Tax of Constrained Decoding in LLM Reflection","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04852","citing_title":"Strengthening Human-Centric Chain-of-Thought Reasoning Integrity in LLMs via a Structured Prompt Framework","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IXC5F7PDSFYDQVS7IOM4RZQ44W","json":"https://pith.science/pith/IXC5F7PDSFYDQVS7IOM4RZQ44W.json","graph_json":"https://pith.science/api/pith-number/IXC5F7PDSFYDQVS7IOM4RZQ44W/graph.json","events_json":"https://pith.science/api/pith-number/IXC5F7PDSFYDQVS7IOM4RZQ44W/events.json","paper":"https://pith.science/paper/IXC5F7PD"},"agent_actions":{"view_html":"https://pith.science/pith/IXC5F7PDSFYDQVS7IOM4RZQ44W","download_json":"https://pith.science/pith/IXC5F7PDSFYDQVS7IOM4RZQ44W.json","view_paper":"https://pith.science/paper/IXC5F7PD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.07919&json=true","fetch_graph":"https://pith.science/api/pith-number/IXC5F7PDSFYDQVS7IOM4RZQ44W/graph.json","fetch_events":"https://pith.science/api/pith-number/IXC5F7PDSFYDQVS7IOM4RZQ44W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IXC5F7PDSFYDQVS7IOM4RZQ44W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IXC5F7PDSFYDQVS7IOM4RZQ44W/action/storage_attestation","attest_author":"https://pith.science/pith/IXC5F7PDSFYDQVS7IOM4RZQ44W/action/author_attestation","sign_citation":"https://pith.science/pith/IXC5F7PDSFYDQVS7IOM4RZQ44W/action/citation_signature","submit_replication":"https://pith.science/pith/IXC5F7PDSFYDQVS7IOM4RZQ44W/action/replication_record"}},"created_at":"2026-07-05T06:49:58.595290+00:00","updated_at":"2026-07-05T06:49:58.595290+00:00"}