{"as_of":"2026-08-07T03:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3bb75a51e81383ba8bc78377b03d43d4c0ef97cb80813fb18dde03064eb7841b","coverage":[{"denominator":22,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":22,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-09T22:55:19.471307Z","state":"measured"},{"denominator":22,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":22,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2604.21327/citation-record","integrity":"/paper/2604.21327/integrity","json":"/paper/2604.21327/citation-record.json","paper":"/paper/2604.21327"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.16400","last_updated":"2025-06-05T17:59:12Z","snapshot_observed_at":"2026-07-06T21:28:24.644692Z","submitted_at":"2025-05-22T08:50:47Z","title":"AceReason-Nemotron: Advancing Math and Code Reasoning through Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.16400","doi":"10.48550/arxiv.2505.16400","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16400","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Acereason-nemotron: Advancing math and code reasoning through reinforcement learning","venue":"ArXiv.org","work_id":"428ad314-c120-41da-9db7-b8bc1918fffb","year":2025},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2505.16400","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:b6eae163ea4d6b67fa893132b70037ec5717b11f99befdedbf21b05c007672d5","observation_id":"993bcdb0-e451-4784-979e-d0e3b9e60613","resolution":{"observed_at":"2026-05-11T14:16:07.278352Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:8e126c758e15766ab61ca01a2ac91d63aa20a33734185eae8ca24bce84a1bdbe","observation_id":"853382d8-ffd6-40b4-895e-ebe07bbe1ab0","resolution":{"observed_at":"2026-05-11T14:16:07.310219Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":"2412.16720","doi":"10.48550/arxiv.2412.16720","metadata_source":"pith","pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"OpenAI o1 System Card","venue":"cs.AI","work_id":"68d3c334-0fc9-49e3-b7b0-a69afae933e2","year":2024},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:2095b6707a45bda370e1e4f37277a974eb4a4f5917d7cff6cce085808e1d20c5","observation_id":"7e8e9d1e-c1e8-4ce8-96c1-20ca7ff6416e","resolution":{"observed_at":"2026-05-11T14:16:07.142595Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15124","last_updated":"2025-04-14T22:39:09Z","snapshot_observed_at":"2026-07-06T19:55:37.400185Z","submitted_at":"2024-11-22T18:44:04Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","version":5},"cited_work":{"arxiv_id":"2411.15124","doi":"10.48550/arxiv.2411.15124","metadata_source":"pith","pith_arxiv_id":"2411.15124","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","venue":"cs.CL","work_id":"28c9dbea-056a-48c2-8000-85f809827e45","year":2024},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2411.15124","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:52a64b159f41e362b8fff961305d3e8ea33bddfa64c463e8aedf37331b917a9c","observation_id":"b004abd2-4277-4aac-99f7-ca41408c70fb","resolution":{"observed_at":"2026-05-11T14:16:07.154064Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.11356","last_updated":"2025-08-29T08:04:20Z","snapshot_observed_at":"2026-08-05T20:04:00.659293Z","submitted_at":"2025-08-15T09:49:14Z","title":"ETTRL: Balancing Exploration and Exploitation in LLM Test-Time Reinforcement Learning Via Entropy Mechanism","version":2},"cited_work":{"arxiv_id":"2508.11356","doi":null,"metadata_source":"pith","pith_arxiv_id":"2508.11356","snapshot_observed_at":"2026-07-10T12:07:03.473290Z","title":"Ettrl: Balancing exploration and exploitation in llm test-time reinforcement learning via entropy mechanism","venue":"cs.LG","work_id":"61f883f4-4072-4ec3-a7b2-f8139ee1feb2","year":2025},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2508.11356","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:e5181961c1977b6e5199dbac0daac306f3c39d87a1965375f2de7eb448381c8c","observation_id":"9a1177b7-7814-4e81-a5e1-4ebf023f189a","resolution":{"observed_at":"2026-05-11T14:16:07.228056Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.22660","last_updated":"2025-06-27T17:25:28Z","snapshot_observed_at":"2026-07-06T21:32:23.537939Z","submitted_at":"2025-05-28T17:59:37Z","title":"Maximizing Confidence Alone Improves Reasoning","version":4},"cited_work":{"arxiv_id":"2505.22660","doi":"10.48550/arxiv.2505.22660","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.22660","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Maximizing confidence alone improves reasoning","venue":"ArXiv.org","work_id":"471b6786-7627-4021-a6b3-7f5cd8c83643","year":2025},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2505.22660","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:c6b7e3832cbf86b98d456749d042a3a3726896d17a4bba711fc2b11033cf58c1","observation_id":"7838304b-0556-4016-b531-a576edb2b8a6","resolution":{"observed_at":"2026-05-11T14:16:07.165280Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":"2412.15115","doi":"10.1145/3581783.3612503","metadata_source":"pith","pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5 Technical Report","venue":"cs.CL","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","year":2024},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:9b62b035c3796b9bc6ca11878d03876dfc079b4bd6965e5172fa5bfb4bd1faaa","observation_id":"53c6ef3a-5f2d-43e1-8940-52451893aa05","resolution":{"observed_at":"2026-05-11T14:16:07.195702Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:48aed9c62c47186559fe0311b9f355cc92d2362c43cdaca9b9dfc7b645ac8021","observation_id":"22c26f1a-5272-4615-b067-806161e7cc74","resolution":{"observed_at":"2026-05-11T14:16:07.358998Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.19256","last_updated":"2024-10-02T04:01:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-28T06:20:03Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","version":2},"cited_work":{"arxiv_id":"2409.19256","doi":"10.1145/3689031.3696075.url:","metadata_source":"pith","pith_arxiv_id":"2409.19256","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","venue":"cs.LG","work_id":"7eb9c9f4-b322-4bba-8011-09ff8d6ad801","year":2024},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2409.19256","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:7828812ee12e80bee03b4dee163ffcc5f18284b9355611f73d9b24c667114899","observation_id":"4ada04d9-6c8e-46a0-91d7-b097450ce33d","resolution":{"observed_at":"2026-05-11T14:16:07.329408Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.22453","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T20:56:13.707156Z","title":"Associated with the WaltonFuture GeoQA-8K-direct-synthesizing dataset release","venue":null,"work_id":"0b4d9d50-bd06-4aac-8a6e-519967317bf2","year":2025},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:b4e6ae2615f8009df807695d988aadcd90b1983858550d7e9ca1a48d54a9aa59","observation_id":"3272d8e9-62e4-4bc8-95aa-6b7d80aaa11b","resolution":{"observed_at":"2026-05-11T14:16:07.183229Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.17938","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T02:26:27.245864Z","title":"Dong Yan, Gaochen Wu, and Bowen Zhou","venue":null,"work_id":"983343c2-e276-4116-ba10-bc37ebec58b1","year":2002},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:2348aacf1b9023bc7c81cab599c49384f57cee792aaf0802ab7952dfef9819ad","observation_id":"ebf84b02-b72e-442b-848c-bd800b138f27","resolution":{"observed_at":"2026-05-11T14:16:07.336008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12122","last_updated":"2024-09-18T16:45:37Z","snapshot_observed_at":"2026-07-06T19:17:41.512834Z","submitted_at":"2024-09-18T16:45:37Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","version":1},"cited_work":{"arxiv_id":"2409.12122","doi":"10.18653/v1/2025.emnlp-main.712","metadata_source":"pith","pith_arxiv_id":"2409.12122","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","venue":"cs.CL","work_id":"a097c5d4-6d32-46ee-9826-57d532bbfc9c","year":2024},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2409.12122","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:67909af8f7dbc829636245c7af21ac1bc3cd0a0356121295e1039808a07be24e","observation_id":"a2c8f175-abc0-4a88-ac49-bc7f09d574ca","resolution":{"observed_at":"2026-05-11T14:16:07.343427Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.03133","last_updated":"2023-07-06T16:59:53Z","snapshot_observed_at":"2026-08-02T16:23:07.138199Z","submitted_at":"2023-07-06T16:59:53Z","title":"Benchmarking Test-Time Adaptation against Distribution Shifts in Image Classification","version":1},"cited_work":{"arxiv_id":"2307.03133","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2307.03133","snapshot_observed_at":"2026-07-01T21:56:16.050107Z","title":"Yongcan Yu, Lijun Sheng, Ran He, and Jian Liang","venue":null,"work_id":"8a0d01be-a4ef-4d1b-8c6f-19e620a398b4","year":2023},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2307.03133","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:63b5e1f2096403a2c2042f1a6a9525a2be14ce42f2ff2c6ade6f2bfa5c2a511e","observation_id":"df0842d8-0d3e-4837-bc14-c47570cad3ec","resolution":{"observed_at":"2026-05-11T14:16:07.318067Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.22271","last_updated":"2025-05-28T11:57:46Z","snapshot_observed_at":"2026-07-06T21:32:09.192939Z","submitted_at":"2025-05-28T11:57:46Z","title":"Test-Time Immunization: A Universal Defense Framework Against Jailbreaks for (Multimodal) Large Language Models","version":1},"cited_work":{"arxiv_id":"2505.22271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.22271","snapshot_observed_at":"2026-06-30T07:24:22.504297Z","title":"arXiv preprint arXiv:2505.22271 (2025)","venue":null,"work_id":"39a4ba8a-81e0-4950-a2ce-4b3d6694f6fc","year":2025},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2505.22271","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:d48a16a2ab3ebca7c369358cb378a6416e4579d8587c30269e0ed4baa5f74694","observation_id":"1b1d1d6c-e1c3-4414-b792-8952e4fd56f8","resolution":{"observed_at":"2026-05-11T14:16:07.293158Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13837","last_updated":"2025-11-24T06:11:04Z","snapshot_observed_at":"2026-07-06T21:11:34.701779Z","submitted_at":"2025-04-18T17:59:56Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","version":5},"cited_work":{"arxiv_id":"2504.13837","doi":"10.48550/arxiv.2504.13837","metadata_source":"pith","pith_arxiv_id":"2504.13837","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","venue":"cs.AI","work_id":"d854765a-e664-41c0-8655-21c4bf2e0cc4","year":2025},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2504.13837","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:896aa2484e4925220390514e572f1955745a65df0fd419e9031a4ce7cd3e8248","observation_id":"40296b30-7cb6-48b9-a5ba-682c59888317","resolution":{"observed_at":"2026-05-11T14:16:07.352061Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:42.002839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:42.002839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.04371","last_updated":"2026-05-21T08:45:38Z","snapshot_observed_at":"2026-07-06T16:03:58.984997Z","submitted_at":"2023-08-08T16:18:20Z","title":"Cumulative Reasoning with Large Language Models","version":11},"cited_work":{"arxiv_id":"2308.04371","doi":null,"metadata_source":"pith","pith_arxiv_id":"2308.04371","snapshot_observed_at":"2026-07-04T08:39:42.734483Z","title":"Cumulative Reasoning with Large Language Models","venue":"cs.AI","work_id":"6fbb9221-8668-4f53-bf1b-90d87cf47a75","year":2023},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2308.04371","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:d317f9946adc552fd4f03f359e650d9cd965b4d1ad54aa4b98f128c617206169","observation_id":"98c45d61-939f-4e99-8b5e-db97e61b2730","resolution":{"observed_at":"2026-05-15T02:43:16.267428Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19590","last_updated":"2026-05-16T23:13:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-26T07:01:06Z","title":"Learning to Reason without External Rewards","version":5},"cited_work":{"arxiv_id":"2505.19590","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.19590","snapshot_observed_at":"2026-07-10T12:07:03.461827Z","title":"Learning to Reason without External Rewards","venue":"cs.LG","work_id":"7286f998-80ea-4c9e-8736-a39c53696142","year":2025},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2505.19590","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:4536ea6cb0d23f17d28ca6be31aefae1d63f1e8cece690abfd3f6c5e01ae2b01","observation_id":"47ac53db-7c7a-400f-857d-031e7774a745","resolution":{"observed_at":"2026-05-15T21:16:57.297836Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16084","last_updated":"2025-06-30T15:59:26Z","snapshot_observed_at":"2026-07-06T21:13:13.686703Z","submitted_at":"2025-04-22T17:59:56Z","title":"TTRL: Test-Time Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2504.16084","doi":"10.48550/arxiv.2504.16084","metadata_source":"pith","pith_arxiv_id":"2504.16084","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"TTRL: Test-Time Reinforcement Learning","venue":"cs.CL","work_id":"54ef1983-e8f7-4b17-9b74-6155b56c8b00","year":2025},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"cited_paper":"/paper/2504.16084","citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:55177658aef389a50f9e8a424dac0b9bf92570954136fd9d2ee5018e3bf76784","observation_id":"02984892-d7eb-4381-9052-b90a5366d86d","resolution":{"observed_at":"2026-05-11T14:16:07.213973Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"28290ba4-d1b7-43c0-89c5-1d7516ff5ff3","year":1983},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:aa3211bd8ae012f811dfe3fbd35657169a03bce016bc2f9e35444bba736362ba","observation_id":"636bcd7c-dbdc-41d3-9723-ce41cf498bfe","resolution":{"observed_at":"2026-05-23T13:58:02.718328Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This confirms that the key assump- tion behind BCS is not specific to MATH-500 but generalizes to other datasets","venue":null,"work_id":"26017a00-fbdd-4989-ba68-037cef1a7779","year":2024},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:209721bd48b3cbbeb71e46544133a9a5a6b5e55940955592c7552fbbd7a64b83","observation_id":"a352f429-65b4-483a-8dec-1fb65aad2099","resolution":{"observed_at":"2026-05-23T13:58:02.714862Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"To investigate the impact of the refinement size on model performance, we conduct a sensitivity analysis by scaling the original setting (1x) to 2x and 4x Table","venue":null,"work_id":"6b668488-328d-4d31-9cd4-ed881f0947ab","year":2024},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:4282879b544c14f945698f41bffe5b86d3d463d108747564743efeb0adcb030c","observation_id":"cf37b11c-444c-4623-b886-3433621892fb","resolution":{"observed_at":"2026-05-23T13:58:02.707784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling up the refinement size yields highly marginal improvements across all benchmarks","venue":null,"work_id":"dbbd01ea-5c1a-48d0-964c-6b5d48a2df89","year":2024},"citing_paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-09T22:55:19.471307Z"},"links":{"citing_paper":"/paper/2604.21327"},"observation_digest":"sha256:0823437695f0f5b37eed8e2172b0b74fc080873594d19b40de8b02a186833571","observation_id":"377fa298-9343-4b83-9f0b-a11375230cce","resolution":{"observed_at":"2026-05-23T13:58:02.711142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.21327","last_updated":"2026-04-23T06:32:08Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:32:08Z","title":"Understanding and Mitigating Spurious Signal Amplification in Test-Time Reinforcement Learning for Math Reasoning"},"reference_resolution":{"displayed":22,"state_counts":{"malformed_identifier":0,"metadata_mismatch":13,"parse_uncertain":0,"unresolved":1,"verified_exact":5,"verified_fuzzy":3},"total_outbound_references":22},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 22 of 22 outbound references and 0 inbound Pith citation observations for arXiv:2604.21327."}