{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IOT23PWLN5EIJJ7LHAONBVLXZW","short_pith_number":"pith:IOT23PWL","schema_version":"1.0","canonical_sha256":"43a7adbecb6f4884a7eb381cd0d577cdb4999c73fc28529ecbf46e59e1e08343","source":{"kind":"arxiv","id":"2411.00750","version":2},"attestation_state":"computed","paper":{"title":"Mitigating Tail Narrowing in LLM Self-Improvement via Socratic-Guided Sampling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Qi Zhang, Tao Gui, Wei He, Xiaowei Shi, Xuanjing Huang, Xunliang Cai, Yitao Zhai, Yiwen Ding, Zhiheng Xi, Zhuoyuan Li","submitted_at":"2024-11-01T17:18:45Z","abstract_excerpt":"Self-improvement methods enable large language models (LLMs) to generate solutions themselves and iteratively train on filtered, high-quality rationales. This process proves effective and reduces the reliance on human supervision in LLMs' reasoning, but the performance soon plateaus. We delve into the process and find that models tend to over-sample on easy queries and under-sample on queries they have yet to master. As iterations proceed, this imbalance in sampling is exacerbated, leading to a long-tail distribution where solutions to difficult queries almost diminish. This phenomenon limits "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.00750","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-01T17:18:45Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3393f4a7d1fe7f30d11deb5cb170e02dd5161ba1e0015ae9c1d903bbfe3c6fc1","abstract_canon_sha256":"c1b78ac7300d9e5c764bebd1c2247ea32c55d07f9a08ae57b943d500e5f34b95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:51.348739Z","signature_b64":"8xauyOqJAEHlh4tbzoZGYSuQILZQrxEzh3NhKzx/Y0KFPAmZRDmvr5UjJfilx2pGLSluitmzslo3iyd/DiHwAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43a7adbecb6f4884a7eb381cd0d577cdb4999c73fc28529ecbf46e59e1e08343","last_reissued_at":"2026-07-05T10:17:51.348281Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:51.348281Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mitigating Tail Narrowing in LLM Self-Improvement via Socratic-Guided Sampling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Qi Zhang, Tao Gui, Wei He, Xiaowei Shi, Xuanjing Huang, Xunliang Cai, Yitao Zhai, Yiwen Ding, Zhiheng Xi, Zhuoyuan Li","submitted_at":"2024-11-01T17:18:45Z","abstract_excerpt":"Self-improvement methods enable large language models (LLMs) to generate solutions themselves and iteratively train on filtered, high-quality rationales. This process proves effective and reduces the reliance on human supervision in LLMs' reasoning, but the performance soon plateaus. We delve into the process and find that models tend to over-sample on easy queries and under-sample on queries they have yet to master. As iterations proceed, this imbalance in sampling is exacerbated, leading to a long-tail distribution where solutions to difficult queries almost diminish. This phenomenon limits "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.00750","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.00750/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.00750","created_at":"2026-07-05T10:17:51.348340+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.00750v2","created_at":"2026-07-05T10:17:51.348340+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.00750","created_at":"2026-07-05T10:17:51.348340+00:00"},{"alias_kind":"pith_short_12","alias_value":"IOT23PWLN5EI","created_at":"2026-07-05T10:17:51.348340+00:00"},{"alias_kind":"pith_short_16","alias_value":"IOT23PWLN5EIJJ7L","created_at":"2026-07-05T10:17:51.348340+00:00"},{"alias_kind":"pith_short_8","alias_value":"IOT23PWL","created_at":"2026-07-05T10:17:51.348340+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.03847","citing_title":"DVPO: Distributional Value Modeling-based Policy Optimization for LLM Post-Training","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04066","citing_title":"Adapt to Thrive! Adaptive Power-Mean Policy Optimization for Improved LLM Reasoning","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04065","citing_title":"Free Energy-Driven Reinforcement Learning with Adaptive Advantage Shaping for Unsupervised Reasoning in LLMs","ref_index":100,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IOT23PWLN5EIJJ7LHAONBVLXZW","json":"https://pith.science/pith/IOT23PWLN5EIJJ7LHAONBVLXZW.json","graph_json":"https://pith.science/api/pith-number/IOT23PWLN5EIJJ7LHAONBVLXZW/graph.json","events_json":"https://pith.science/api/pith-number/IOT23PWLN5EIJJ7LHAONBVLXZW/events.json","paper":"https://pith.science/paper/IOT23PWL"},"agent_actions":{"view_html":"https://pith.science/pith/IOT23PWLN5EIJJ7LHAONBVLXZW","download_json":"https://pith.science/pith/IOT23PWLN5EIJJ7LHAONBVLXZW.json","view_paper":"https://pith.science/paper/IOT23PWL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.00750&json=true","fetch_graph":"https://pith.science/api/pith-number/IOT23PWLN5EIJJ7LHAONBVLXZW/graph.json","fetch_events":"https://pith.science/api/pith-number/IOT23PWLN5EIJJ7LHAONBVLXZW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IOT23PWLN5EIJJ7LHAONBVLXZW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IOT23PWLN5EIJJ7LHAONBVLXZW/action/storage_attestation","attest_author":"https://pith.science/pith/IOT23PWLN5EIJJ7LHAONBVLXZW/action/author_attestation","sign_citation":"https://pith.science/pith/IOT23PWLN5EIJJ7LHAONBVLXZW/action/citation_signature","submit_replication":"https://pith.science/pith/IOT23PWLN5EIJJ7LHAONBVLXZW/action/replication_record"}},"created_at":"2026-07-05T10:17:51.348340+00:00","updated_at":"2026-07-05T10:17:51.348340+00:00"}