{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3UB3FNSSA5GICDFOPQIUQGDJFW","short_pith_number":"pith:3UB3FNSS","schema_version":"1.0","canonical_sha256":"dd03b2b652074c810cae7c114818692d8e4630bf3c14fe70bb4cffefceef5897","source":{"kind":"arxiv","id":"2311.07961","version":1},"attestation_state":"computed","paper":{"title":"The ART of LLM Refinement: Ask, Refine, and Trust","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andrew Cohen, Asli Celikyilmaz, Jason Weston, Koustuv Sinha, Kumar Shridhar, Mrinmaya Sachan, Ping Yu, Ram Pasunuru, Tianlu Wang","submitted_at":"2023-11-14T07:26:32Z","abstract_excerpt":"In recent years, Large Language Models (LLMs) have demonstrated remarkable generative abilities, but can they judge the quality of their own generations? A popular concept, referred to as self-refinement, postulates that LLMs can detect and correct the errors in their generations when asked to do so. However, recent empirical evidence points in the opposite direction, suggesting that LLMs often struggle to accurately identify errors when reasoning is involved. To address this, we propose a reasoning with refinement objective called ART: Ask, Refine, and Trust, which asks necessary questions to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.07961","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-14T07:26:32Z","cross_cats_sorted":[],"title_canon_sha256":"f76bd59c1cc7028d59a0a5369cce4de5ec3f36ebe44a4e83bfd7352a4ca92769","abstract_canon_sha256":"69bcadffd418172598ee1c6f94258becc26157254dac97633932227c4570503b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:26.393071Z","signature_b64":"ZziF1C0bq3Fl/s0g4iu3pFI9N7/U7sv3R2Coaa7PIwaVCQ3z2KeE7FuiCJX8aFuCC3as2RjvpKX9Aj/CopfLBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd03b2b652074c810cae7c114818692d8e4630bf3c14fe70bb4cffefceef5897","last_reissued_at":"2026-07-05T07:12:26.392611Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:26.392611Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The ART of LLM Refinement: Ask, Refine, and Trust","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andrew Cohen, Asli Celikyilmaz, Jason Weston, Koustuv Sinha, Kumar Shridhar, Mrinmaya Sachan, Ping Yu, Ram Pasunuru, Tianlu Wang","submitted_at":"2023-11-14T07:26:32Z","abstract_excerpt":"In recent years, Large Language Models (LLMs) have demonstrated remarkable generative abilities, but can they judge the quality of their own generations? A popular concept, referred to as self-refinement, postulates that LLMs can detect and correct the errors in their generations when asked to do so. However, recent empirical evidence points in the opposite direction, suggesting that LLMs often struggle to accurately identify errors when reasoning is involved. To address this, we propose a reasoning with refinement objective called ART: Ask, Refine, and Trust, which asks necessary questions to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.07961","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.07961/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.07961","created_at":"2026-07-05T07:12:26.392672+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.07961v1","created_at":"2026-07-05T07:12:26.392672+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.07961","created_at":"2026-07-05T07:12:26.392672+00:00"},{"alias_kind":"pith_short_12","alias_value":"3UB3FNSSA5GI","created_at":"2026-07-05T07:12:26.392672+00:00"},{"alias_kind":"pith_short_16","alias_value":"3UB3FNSSA5GICDFO","created_at":"2026-07-05T07:12:26.392672+00:00"},{"alias_kind":"pith_short_8","alias_value":"3UB3FNSS","created_at":"2026-07-05T07:12:26.392672+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.24858","citing_title":"Context-Mediated Domain Adaptation in Multi-Agent Sensemaking Systems","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3UB3FNSSA5GICDFOPQIUQGDJFW","json":"https://pith.science/pith/3UB3FNSSA5GICDFOPQIUQGDJFW.json","graph_json":"https://pith.science/api/pith-number/3UB3FNSSA5GICDFOPQIUQGDJFW/graph.json","events_json":"https://pith.science/api/pith-number/3UB3FNSSA5GICDFOPQIUQGDJFW/events.json","paper":"https://pith.science/paper/3UB3FNSS"},"agent_actions":{"view_html":"https://pith.science/pith/3UB3FNSSA5GICDFOPQIUQGDJFW","download_json":"https://pith.science/pith/3UB3FNSSA5GICDFOPQIUQGDJFW.json","view_paper":"https://pith.science/paper/3UB3FNSS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.07961&json=true","fetch_graph":"https://pith.science/api/pith-number/3UB3FNSSA5GICDFOPQIUQGDJFW/graph.json","fetch_events":"https://pith.science/api/pith-number/3UB3FNSSA5GICDFOPQIUQGDJFW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3UB3FNSSA5GICDFOPQIUQGDJFW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3UB3FNSSA5GICDFOPQIUQGDJFW/action/storage_attestation","attest_author":"https://pith.science/pith/3UB3FNSSA5GICDFOPQIUQGDJFW/action/author_attestation","sign_citation":"https://pith.science/pith/3UB3FNSSA5GICDFOPQIUQGDJFW/action/citation_signature","submit_replication":"https://pith.science/pith/3UB3FNSSA5GICDFOPQIUQGDJFW/action/replication_record"}},"created_at":"2026-07-05T07:12:26.392672+00:00","updated_at":"2026-07-05T07:12:26.392672+00:00"}