{"as_of":"2026-08-07T04:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5b93bcd4e559d6a1a73e7e0b8c5eaabb5488e5c3a7751818f2476ee1ded07ae4","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":24,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T23:35:24.315457Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T04:19:34.020533Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-08-06T23:35:24.315457Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.17219","last_updated":"2025-06-25T13:27:49Z","snapshot_observed_at":"2026-08-06T23:28:29.552211Z","submitted_at":"2025-06-20T17:59:52Z","title":"No Free Lunch: Rethinking Internal Feedback for LLM Reasoning","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T23:35:24.315457Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2506.17219"},"observation_digest":"sha256:235b07f6cb002ad8b5e49e8d795940385db9b8e3a6c6e44c5c7156e28e5dfc32","observation_id":"a7b0ad1c-dfe7-4e4a-adae-ca65f8b91856","resolution":{"observed_at":"2026-08-06T23:35:24.315457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-08-06T14:43:52.966250Z","title":"Confidence is all you need: Few-shot rl fine-tuning of language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.18122","last_updated":"2025-07-24T06:17:39Z","snapshot_observed_at":"2026-08-06T14:36:16.245142Z","submitted_at":"2025-07-24T06:17:39Z","title":"Maximizing Prefix-Confidence at Test-Time Efficiently Improves Mathematical Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T14:43:52.966250Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2507.18122"},"observation_digest":"sha256:d0a87725ab2f47cd082a080ce986b623f1575c4b38b0ac4cf0a1e894e54b2d6c","observation_id":"bfc426df-37f3-4055-877d-698b692486d3","resolution":{"observed_at":"2026-08-06T14:43:52.966250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-08-06T15:38:05.011922Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"reference_index":280,"source":"arxiv_source","source_observed_at":"2026-05-18T00:02:24.352947Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2509.08827"},"observation_digest":"sha256:e80b186a11131305ccffbcc4202a6d969e8f31d01a62e555e842d34a955bdf4f","observation_id":"f726fd13-53e2-436d-a482-265d307afd9b","resolution":{"observed_at":"2026-05-18T00:02:24.731798Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2509.14234","last_updated":"2026-05-09T12:06:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-17T17:59:42Z","title":"Compute as Teacher: Turning Inference Compute Into Reference-Free Supervision","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-18T15:45:09.730804Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2509.14234"},"observation_digest":"sha256:4aa51e9cf450abba00f351d3591682237176948c6752a6d1b183a1fa112ac0b8","observation_id":"0b366e46-e0ea-4246-8b46-8beaf67c8df4","resolution":{"observed_at":"2026-05-18T15:46:34.049081Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-08-04T16:07:33.243783Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.16679","last_updated":"2025-09-20T13:11:28Z","snapshot_observed_at":"2026-08-04T16:07:24.699834Z","submitted_at":"2025-09-20T13:11:28Z","title":"Reinforcement Learning Meets Large Language Models: A Survey of Advancements and Applications Across the LLM Lifecycle","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-04T16:07:33.243783Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2509.16679"},"observation_digest":"sha256:a329dbc77b8043fe1eec685f63fa805ad091ffec050aea2918237ac4411100f0","observation_id":"0657594b-20a4-42e4-84e5-f1d2760997d2","resolution":{"observed_at":"2026-08-04T16:07:33.243783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-08-04T13:42:31.638150Z","title":"AGIQA-3K: An open database for AI-generated image quality assessment.IEEE Transactions on Circuits and Systems for Video Technology, 34:6833–6846, 2023a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.25787","last_updated":"2026-06-11T06:39:23Z","snapshot_observed_at":"2026-08-04T13:42:31.049083Z","submitted_at":"2025-09-30T04:57:26Z","title":"Self-Evolving Vision-Language Models for Image Quality Assessment via Voting and Ranking","version":5},"reference_index":2010,"source":"pdf_text","source_observed_at":"2026-08-04T13:42:31.638150Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2509.25787"},"observation_digest":"sha256:4fd9aa5a06fc62de7a39714eaf4517babcf5b303257883b6653014168c976afe","observation_id":"5c31a09d-140a-4ba0-b9d4-c89ed43e3847","resolution":{"observed_at":"2026-08-04T13:42:31.638150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-08-04T10:44:30.709604Z","title":"Confidence is all you need: Few-shot rl fine-tuning of language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.08977","last_updated":"2026-06-02T08:25:17Z","snapshot_observed_at":"2026-08-06T09:42:32.578654Z","submitted_at":"2025-10-10T03:38:17Z","title":"Breaking the Self-Confirming Loop: Diagnosing and Mitigating Systemic Reward Bias in Self-Rewarding RL","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-04T10:44:30.709604Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2510.08977"},"observation_digest":"sha256:874da2c30cfa51fce225ec3d3fb93765c7ce8aa8ab95cc4309792063722cd359","observation_id":"eb0f7f68-99bc-46f6-9fd1-a97d1f67acd1","resolution":{"observed_at":"2026-08-04T10:44:30.709604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-08-03T05:14:19.263434Z","title":"Confidence is all you need: Few-shot rl fine-tuning of language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.02979","last_updated":"2026-05-24T21:03:36Z","snapshot_observed_at":"2026-08-03T05:14:15.579080Z","submitted_at":"2026-02-03T01:38:53Z","title":"CPMobius: Iterative Coach-Player Reasoning for Data-Free Reinforcement Learning","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-03T05:14:19.263434Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2602.02979"},"observation_digest":"sha256:eb4cc31c9492b2570c4d337f4df0015f3a5054b2f8c581633d3d506f2082a23c","observation_id":"80755571-d9f4-4a77-9222-4be56575688e","resolution":{"observed_at":"2026-08-03T05:14:19.263434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2603.09117","last_updated":"2026-05-27T02:49:05Z","snapshot_observed_at":"2026-07-15T12:13:14.577374Z","submitted_at":"2026-03-10T02:47:59Z","title":"Decoupling Reasoning and Confidence: Resurrecting Calibration in Reinforcement Learning from Verifiable Rewards","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-15T13:49:11.758336Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2603.09117"},"observation_digest":"sha256:0a2f56ba0627d0120ec3ab3abf26a4c22c9aa8864fbd6c207b2bb40723442c0b","observation_id":"0a48785a-3065-4a6b-a1a4-f3a9b0ac43cd","resolution":{"observed_at":"2026-05-15T13:50:02.385698Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-14T20:22:12.729190Z","title":"Li, P., Skripkin, M., Zubrey, A., Kuznetsov, A., and Oseledets, I","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18375","last_updated":"2026-07-10T23:41:27Z","snapshot_observed_at":"2026-08-04T02:39:29.405773Z","submitted_at":"2026-03-19T00:23:57Z","title":"Relationship-Centered Care: Relatedness and Responsible Design for Human Connections in Mental-Health Care","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-14T20:22:12.729190Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2603.18375"},"observation_digest":"sha256:6c7ee20715752c4a9014afeae74c299b8ce42026661b00ef591cc479099b9f77","observation_id":"7a5f517b-1ed4-42c9-ab8b-c8c24149ceb9","resolution":{"observed_at":"2026-07-14T20:22:12.729190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2604.03993","last_updated":"2026-04-05T06:30:50Z","snapshot_observed_at":"2026-07-06T22:53:01.835843Z","submitted_at":"2026-04-05T06:30:50Z","title":"Can LLMs Learn to Reason Robustly under Noisy Supervision?","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-13T16:58:42.129870Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2604.03993"},"observation_digest":"sha256:2ceb9cd600caabb8cfec209ea38c9126a252131654e00a30915c47575b40b353","observation_id":"5b8e06bf-4516-44e3-9dd5-59aa413d6e3f","resolution":{"observed_at":"2026-05-13T17:08:01.328397Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2604.17928","last_updated":"2026-04-20T08:09:01Z","snapshot_observed_at":"2026-07-06T23:04:55.190916Z","submitted_at":"2026-04-20T08:09:01Z","title":"HEALing Entropy Collapse: Enhancing Exploration in Few-Shot RLVR via Hybrid-Domain Entropy Dynamics Alignment","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-10T05:23:08.478393Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2604.17928"},"observation_digest":"sha256:06e3aedf9ee384049af8a573587b95839bc24b066f09b6668bf588b75429ba22","observation_id":"1b5a1f0a-dbe7-4c35-8427-3e4795804469","resolution":{"observed_at":"2026-05-10T09:23:37.520524Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2605.01428","last_updated":"2026-05-02T12:59:14Z","snapshot_observed_at":"2026-07-06T23:14:38.379875Z","submitted_at":"2026-05-02T12:59:14Z","title":"Hallucinations Undermine Trust; Metacognition is a Way Forward","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-09T14:29:54.924293Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2605.01428"},"observation_digest":"sha256:37850dea5a61603d3785f8eb1dffe2d312cd3f1de20752dc86bb0555d7f3d652","observation_id":"136bdb10-f5df-4391-bbbf-5716b98d57f1","resolution":{"observed_at":"2026-05-11T16:56:07.150385Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2605.01853","last_updated":"2026-05-03T12:46:41Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T12:46:41Z","title":"Spatiotemporal Hidden-State Dynamics as a Signature of Internal Reasoning in Large Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-09T17:20:19.586214Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2605.01853"},"observation_digest":"sha256:2ab221b32512ff5d261cfa73ede39d79d79d5a79459925bae2e9e5f1e6e78589","observation_id":"2225a65a-24bb-45d7-918a-647dc569ecf7","resolution":{"observed_at":"2026-05-11T16:21:08.878934Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2605.04065","last_updated":"2026-05-07T04:49:30Z","snapshot_observed_at":"2026-07-06T23:16:54.673178Z","submitted_at":"2026-04-11T07:26:04Z","title":"Free Energy-Driven Reinforcement Learning with Adaptive Advantage Shaping for Unsupervised Reasoning in LLMs","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-05-10T16:58:10.013475Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2605.04065"},"observation_digest":"sha256:2287c2e4fe75cacbe8b28f3ef8768aaadeff839a4cefac3b4a2f4f154e75d638","observation_id":"91676cf1-799a-4ac2-8233-5761a453ac46","resolution":{"observed_at":"2026-05-11T07:45:59.928030Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2605.07244","last_updated":"2026-05-08T05:01:40Z","snapshot_observed_at":"2026-07-06T23:19:35.885430Z","submitted_at":"2026-05-08T05:01:40Z","title":"Experience Sharing in Mutual Reinforcement Learning for Heterogeneous Language Models","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-11T02:02:41.411795Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2605.07244"},"observation_digest":"sha256:9163a839ae0ab34eaa3d449ad3432aafb95ebe7507415d57709922e2b0c4fe5a","observation_id":"65dc5c60-d756-42d2-8f0d-482551526225","resolution":{"observed_at":"2026-05-11T04:00:54.981604Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2605.13467","last_updated":"2026-05-13T12:55:18Z","snapshot_observed_at":"2026-08-03T01:46:31.336725Z","submitted_at":"2026-05-13T12:55:18Z","title":"PDCR: Perception-Decomposed Confidence Reward for Vision-Language Reasoning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-14T19:20:32.435135Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2605.13467"},"observation_digest":"sha256:989757872c09e4a1a5843c86350775817dccf4c774488f19f34e5867f5158f64","observation_id":"0adf9594-fec1-4beb-906f-ffcf2f486f1f","resolution":{"observed_at":"2026-05-14T19:22:50.601594Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2605.15012","last_updated":"2026-05-14T16:12:30Z","snapshot_observed_at":"2026-07-06T23:26:23.179482Z","submitted_at":"2026-05-14T16:12:30Z","title":"Boosting Reinforcement Learning with Verifiable Rewards via Randomly Selected Few-Shot Guidance","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-15T03:18:26.590871Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2605.15012"},"observation_digest":"sha256:56f544797fe2b2b4688c4d4e0d53c62e3f099615f866cd5fd95151640413dc80","observation_id":"0fb3c622-79d3-4020-b0d5-d46f3c71b144","resolution":{"observed_at":"2026-05-15T03:19:43.052901Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2605.19444","last_updated":"2026-05-27T03:12:51Z","snapshot_observed_at":"2026-07-06T23:30:11.337726Z","submitted_at":"2026-05-19T06:58:44Z","title":"Detecting and Mitigating the Correct-Answer Extinction Window in Test-Time Reinforcement Learning with Majority Voting","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-20T07:20:50.835826Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2605.19444"},"observation_digest":"sha256:afc9114fb5412022e59a824e33bfbd1b750a8d37bcce34735de913579c4f4e06","observation_id":"f4b05f7d-a7a9-483f-9ea8-8cd9c4ece2ea","resolution":{"observed_at":"2026-05-20T07:23:06.927905Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2606.01249","last_updated":"2026-06-17T04:44:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-31T14:04:51Z","title":"Trust Region On-Policy Distillation","version":3},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-06-28T17:38:50.313305Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2606.01249"},"observation_digest":"sha256:01b26f493f9b2c676c41beab574773e368c4ad448bd73e3d658cec4aeaf1bd26","observation_id":"c1d205cb-439a-471a-a099-f5e14cd33be0","resolution":{"observed_at":"2026-07-01T20:56:13.714700Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2606.04503","last_updated":"2026-06-03T06:34:42Z","snapshot_observed_at":"2026-07-06T23:44:37.797802Z","submitted_at":"2026-06-03T06:34:42Z","title":"Smart Picks in the Dark: Towards Efficient RLVR for Reasoning via Tracing Metacognitive Pivots","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-28T06:55:09.927034Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2606.04503"},"observation_digest":"sha256:3044ef25c3704c6b08aa20c0655837e1130d286040c38fb7b13f666e151271f3","observation_id":"3fcabd54-4272-4f59-805a-86b1790d9ca2","resolution":{"observed_at":"2026-07-02T07:26:46.294521Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2606.04516","last_updated":"2026-06-03T06:47:50Z","snapshot_observed_at":"2026-07-06T23:44:37.797802Z","submitted_at":"2026-06-03T06:47:50Z","title":"GeoMin: Data-Efficient Semi-Supervised RLVR via Geometric Distribution Modeling","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-06-28T07:45:43.320339Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2606.04516"},"observation_digest":"sha256:4aa5646e22849b5ea4225473642ece0378b5361fd268765c8206d9c22b8b4937","observation_id":"83ccfe0e-b653-4d2c-9dac-b0f2cfa1cb3c","resolution":{"observed_at":"2026-07-02T06:06:40.756454Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2606.11634","last_updated":"2026-06-10T03:56:03Z","snapshot_observed_at":"2026-08-02T13:37:43.737297Z","submitted_at":"2026-06-10T03:56:03Z","title":"Architecture-Aware Reinforcement Learning Makes Sliding-Window Attention Competitive in Math Reasoning","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-27T10:18:54.163862Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2606.11634"},"observation_digest":"sha256:09a66ab699ad9820c27e431c19f4947dc38d99a7f7fc1b204946a8d119b17043","observation_id":"a1e595f4-6dc3-404d-8ca8-e0eea46b1270","resolution":{"observed_at":"2026-07-03T09:47:59.871201Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","version":3},"cited_work":{"arxiv_id":"2506.06395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06395","snapshot_observed_at":"2026-07-04T04:19:34.020533Z","title":"arXiv preprint arXiv:2506.06395 , year=","venue":null,"work_id":"7732e7d1-769d-48e4-975d-770d911ba6be","year":2025},"citing_paper":{"arxiv_id":"2606.20881","last_updated":"2026-06-18T19:15:50Z","snapshot_observed_at":"2026-07-06T23:55:57.945000Z","submitted_at":"2026-06-18T19:15:50Z","title":"When Do Intrinsic Rewards Work for Code Reasoning? A Comprehensive Study","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-26T17:07:21.486960Z"},"links":{"cited_paper":"/paper/2506.06395","citing_paper":"/paper/2606.20881"},"observation_digest":"sha256:ad43b0ccf00376e66d54d41215165ced9b028ca1ed045ee4c97ce72b8eb2e337","observation_id":"bab38903-170f-4d40-b597-a461b8816476","resolution":{"observed_at":"2026-07-04T04:19:34.022356Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.06395/citation-record","integrity":"/paper/2506.06395/integrity","json":"/paper/2506.06395/citation-record.json","paper":"/paper/2506.06395"},"outbound":[],"paper":{"arxiv_id":"2506.06395","last_updated":"2025-06-11T06:21:59Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T03:25:04.843230Z","submitted_at":"2025-06-05T19:55:15Z","title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 24 inbound Pith citation observations for arXiv:2506.06395."}