{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MMQLOE4JGAZ4NSQS5MOTOOUVWT","short_pith_number":"pith:MMQLOE4J","schema_version":"1.0","canonical_sha256":"6320b713893033c6ca12eb1d373a95b4f520a851d4c49795575b51517ba82c45","source":{"kind":"arxiv","id":"2505.00662","version":1},"attestation_state":"computed","paper":{"title":"DeepCritic: Deliberate Critique with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jingwen Chen, Ji-Rong Wen, Wenkai Yang, Yankai Lin","submitted_at":"2025-05-01T17:03:17Z","abstract_excerpt":"As Large Language Models (LLMs) are rapidly evolving, providing accurate feedback and scalable oversight on their outputs becomes an urgent and critical problem. Leveraging LLMs as critique models to achieve automated supervision is a promising solution. In this work, we focus on studying and enhancing the math critique ability of LLMs. Current LLM critics provide critiques that are too shallow and superficial on each step, leading to low judgment accuracy and struggling to offer sufficient feedback for the LLM generator to correct mistakes. To tackle this issue, we propose a novel and effecti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.00662","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-01T17:03:17Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"932ce1da03bab0211db3e4a01efddccf5cb802fcea9b8606118340831593660b","abstract_canon_sha256":"c1bfa6b8c1bab448d74470fb77947a2dc6adb72fc9fd592c0cc6764a6c736c0b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:57:23.639682Z","signature_b64":"c5uVR6b87it8GzQ9hca7bHTFHpkSqprtxZWz5uk9HaOK6FsTH6J7oPPtUWFtnc1NlOpq5LbSB0tOip0NTtRjDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6320b713893033c6ca12eb1d373a95b4f520a851d4c49795575b51517ba82c45","last_reissued_at":"2026-07-05T10:57:23.639196Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:57:23.639196Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DeepCritic: Deliberate Critique with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jingwen Chen, Ji-Rong Wen, Wenkai Yang, Yankai Lin","submitted_at":"2025-05-01T17:03:17Z","abstract_excerpt":"As Large Language Models (LLMs) are rapidly evolving, providing accurate feedback and scalable oversight on their outputs becomes an urgent and critical problem. Leveraging LLMs as critique models to achieve automated supervision is a promising solution. In this work, we focus on studying and enhancing the math critique ability of LLMs. Current LLM critics provide critiques that are too shallow and superficial on each step, leading to low judgment accuracy and struggling to offer sufficient feedback for the LLM generator to correct mistakes. To tackle this issue, we propose a novel and effecti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.00662","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.00662/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.00662","created_at":"2026-07-05T10:57:23.639252+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.00662v1","created_at":"2026-07-05T10:57:23.639252+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.00662","created_at":"2026-07-05T10:57:23.639252+00:00"},{"alias_kind":"pith_short_12","alias_value":"MMQLOE4JGAZ4","created_at":"2026-07-05T10:57:23.639252+00:00"},{"alias_kind":"pith_short_16","alias_value":"MMQLOE4JGAZ4NSQS","created_at":"2026-07-05T10:57:23.639252+00:00"},{"alias_kind":"pith_short_8","alias_value":"MMQLOE4J","created_at":"2026-07-05T10:57:23.639252+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22600","citing_title":"On the Position Bias of On-Policy Distillation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22600","citing_title":"On the Position Bias of On-Policy Distillation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15113","citing_title":"Learning from Language Feedback via Variational Policy Distillation","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MMQLOE4JGAZ4NSQS5MOTOOUVWT","json":"https://pith.science/pith/MMQLOE4JGAZ4NSQS5MOTOOUVWT.json","graph_json":"https://pith.science/api/pith-number/MMQLOE4JGAZ4NSQS5MOTOOUVWT/graph.json","events_json":"https://pith.science/api/pith-number/MMQLOE4JGAZ4NSQS5MOTOOUVWT/events.json","paper":"https://pith.science/paper/MMQLOE4J"},"agent_actions":{"view_html":"https://pith.science/pith/MMQLOE4JGAZ4NSQS5MOTOOUVWT","download_json":"https://pith.science/pith/MMQLOE4JGAZ4NSQS5MOTOOUVWT.json","view_paper":"https://pith.science/paper/MMQLOE4J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.00662&json=true","fetch_graph":"https://pith.science/api/pith-number/MMQLOE4JGAZ4NSQS5MOTOOUVWT/graph.json","fetch_events":"https://pith.science/api/pith-number/MMQLOE4JGAZ4NSQS5MOTOOUVWT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MMQLOE4JGAZ4NSQS5MOTOOUVWT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MMQLOE4JGAZ4NSQS5MOTOOUVWT/action/storage_attestation","attest_author":"https://pith.science/pith/MMQLOE4JGAZ4NSQS5MOTOOUVWT/action/author_attestation","sign_citation":"https://pith.science/pith/MMQLOE4JGAZ4NSQS5MOTOOUVWT/action/citation_signature","submit_replication":"https://pith.science/pith/MMQLOE4JGAZ4NSQS5MOTOOUVWT/action/replication_record"}},"created_at":"2026-07-05T10:57:23.639252+00:00","updated_at":"2026-07-05T10:57:23.639252+00:00"}