{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2UDZA7K5ILTHKIO2UXMRUVPTWZ","short_pith_number":"pith:2UDZA7K5","schema_version":"1.0","canonical_sha256":"d507907d5d42e67521daa5d91a55f3b66f92b8d22b9cc6be65001ccf0944d17e","source":{"kind":"arxiv","id":"2506.08243","version":1},"attestation_state":"computed","paper":{"title":"Temporalizing Confidence: Evaluation of Chain-of-Thought Reasoning with Signal Temporal Logic","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Artem Bisliouk, Ivan Ruchkin, Rohith Reddy Nama, Zhenjiang Mao","submitted_at":"2025-06-09T21:21:12Z","abstract_excerpt":"Large Language Models (LLMs) have shown impressive performance in mathematical reasoning tasks when guided by Chain-of-Thought (CoT) prompting. However, they tend to produce highly confident yet incorrect outputs, which poses significant risks in domains like education, where users may lack the expertise to assess reasoning steps. To address this, we propose a structured framework that models stepwise confidence as a temporal signal and evaluates it using Signal Temporal Logic (STL). In particular, we define formal STL-based constraints to capture desirable temporal properties and compute robu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.08243","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-09T21:21:12Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"014305815b04c380a510b89b35ef9d276f0780d777e89ba138fc97f9c6f49e0a","abstract_canon_sha256":"b699827950c89f7f3fc0e9fac90ab5c38a1a6c188c4d038de6e71b8ae0d3ad34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:09.071050Z","signature_b64":"cxv/N0KkLMvsVkbqTze7AYMDVXvtF2mUGKk47ZVwvpvlxqHOJrbzIdE/Mlne+/PFOpV7vcbhJPpk6ldlqbNwAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d507907d5d42e67521daa5d91a55f3b66f92b8d22b9cc6be65001ccf0944d17e","last_reissued_at":"2026-07-05T11:19:09.070651Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:09.070651Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Temporalizing Confidence: Evaluation of Chain-of-Thought Reasoning with Signal Temporal Logic","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Artem Bisliouk, Ivan Ruchkin, Rohith Reddy Nama, Zhenjiang Mao","submitted_at":"2025-06-09T21:21:12Z","abstract_excerpt":"Large Language Models (LLMs) have shown impressive performance in mathematical reasoning tasks when guided by Chain-of-Thought (CoT) prompting. However, they tend to produce highly confident yet incorrect outputs, which poses significant risks in domains like education, where users may lack the expertise to assess reasoning steps. To address this, we propose a structured framework that models stepwise confidence as a temporal signal and evaluates it using Signal Temporal Logic (STL). In particular, we define formal STL-based constraints to capture desirable temporal properties and compute robu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.08243","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.08243/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.08243","created_at":"2026-07-05T11:19:09.070703+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.08243v1","created_at":"2026-07-05T11:19:09.070703+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.08243","created_at":"2026-07-05T11:19:09.070703+00:00"},{"alias_kind":"pith_short_12","alias_value":"2UDZA7K5ILTH","created_at":"2026-07-05T11:19:09.070703+00:00"},{"alias_kind":"pith_short_16","alias_value":"2UDZA7K5ILTHKIO2","created_at":"2026-07-05T11:19:09.070703+00:00"},{"alias_kind":"pith_short_8","alias_value":"2UDZA7K5","created_at":"2026-07-05T11:19:09.070703+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.07238","citing_title":"Systematic Optimization of Open Source Large Language Models for Mathematical Reasoning","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2UDZA7K5ILTHKIO2UXMRUVPTWZ","json":"https://pith.science/pith/2UDZA7K5ILTHKIO2UXMRUVPTWZ.json","graph_json":"https://pith.science/api/pith-number/2UDZA7K5ILTHKIO2UXMRUVPTWZ/graph.json","events_json":"https://pith.science/api/pith-number/2UDZA7K5ILTHKIO2UXMRUVPTWZ/events.json","paper":"https://pith.science/paper/2UDZA7K5"},"agent_actions":{"view_html":"https://pith.science/pith/2UDZA7K5ILTHKIO2UXMRUVPTWZ","download_json":"https://pith.science/pith/2UDZA7K5ILTHKIO2UXMRUVPTWZ.json","view_paper":"https://pith.science/paper/2UDZA7K5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.08243&json=true","fetch_graph":"https://pith.science/api/pith-number/2UDZA7K5ILTHKIO2UXMRUVPTWZ/graph.json","fetch_events":"https://pith.science/api/pith-number/2UDZA7K5ILTHKIO2UXMRUVPTWZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2UDZA7K5ILTHKIO2UXMRUVPTWZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2UDZA7K5ILTHKIO2UXMRUVPTWZ/action/storage_attestation","attest_author":"https://pith.science/pith/2UDZA7K5ILTHKIO2UXMRUVPTWZ/action/author_attestation","sign_citation":"https://pith.science/pith/2UDZA7K5ILTHKIO2UXMRUVPTWZ/action/citation_signature","submit_replication":"https://pith.science/pith/2UDZA7K5ILTHKIO2UXMRUVPTWZ/action/replication_record"}},"created_at":"2026-07-05T11:19:09.070703+00:00","updated_at":"2026-07-05T11:19:09.070703+00:00"}