{"as_of":"2026-08-06T14:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:43bf656ef972edb43830b5f3c220c5a34086c8845517c6ef557b79096f14a45f","coverage":[{"denominator":50,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T11:50:26.030339Z","state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.10810/citation-record","integrity":"/paper/2605.10810/integrity","json":"/paper/2605.10810/citation-record.json","paper":"/paper/2605.10810"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tülu 3: Pushing Frontiers in Open Language Model Post-Training","venue":null,"work_id":"0ec05ab4-7832-41ed-ba4c-de76a7af5e3d","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:09c4ad72663ab824229e7d85b58f2be7daadcbf77f0d22ea2f9304584b860b78","observation_id":"61a5cee9-16e9-43e0-bac9-47ee9a802aa8","resolution":{"observed_at":"2026-05-19T17:22:42.172403Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:d182979ace035ddc4173ea2bb331f772c8614ce447efd8761f8b0e2cbbfcfdce","observation_id":"ecdb4662-0742-48f9-925d-de1e0267996e","resolution":{"observed_at":"2026-05-19T17:22:41.954792Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:21e5575ea8dbe41e5ac2d72bb6afa20c3769b69041bbc552f7a75a8b7fe44162","observation_id":"d840f19f-a558-438d-bdcc-3e49d58a35f0","resolution":{"observed_at":"2026-05-19T17:22:41.936034Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"cited_work":{"arxiv_id":"2406.19314","doi":"10.48550/arxiv.2406.19314","metadata_source":"pith","pith_arxiv_id":"2406.19314","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","venue":"cs.CL","work_id":"6b2b33bf-350e-4ee2-b8a7-f011e53384e7","year":2024},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2406.19314","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:cb9c712fa2e1d2937af8522fa7af4829a7ee48ad55c24d56a56af59459da5717","observation_id":"3ead951c-6459-4ee8-a072-3b0502df03ae","resolution":{"observed_at":"2026-05-19T17:22:42.004197Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.acl-long.901","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AntiLeakBench: Preventing Data Contamination by Automatically Con- structing Benchmarks with Updated Real-World Knowledge","venue":null,"work_id":"5985d981-c098-467d-91fa-8ef582e98d2a","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:6bfd5b101cc84d6c6cd0ef05a139579b0cae510b54d299d0e3d4351de299c248","observation_id":"d5124c3e-c4ec-4b63-bc29-25be03d5747a","resolution":{"observed_at":"2026-05-19T17:22:41.656914Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1609/aaai","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T02:04:26.079898Z","title":"Louis, G","venue":null,"work_id":"9f349f1f-0e39-446f-8ac6-694a06c25de5","year":2026},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:483325d6436567584b9fa1b9346cbccc87fe3bb640e0668f635fbef0285ac68d","observation_id":"efed21f2-50f5-4909-9a85-fe123c82a212","resolution":{"observed_at":"2026-05-19T17:22:41.650016Z","resolver_source":"doi_truncated","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.10760","last_updated":"2022-10-19T17:56:10Z","snapshot_observed_at":"2026-07-06T14:07:50.468967Z","submitted_at":"2022-10-19T17:56:10Z","title":"Scaling Laws for Reward Model Overoptimization","version":1},"cited_work":{"arxiv_id":"2210.10760","doi":"10.48550/arxiv.2210.10760","metadata_source":"pith","pith_arxiv_id":"2210.10760","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling Laws for Reward Model Overoptimization","venue":"cs.LG","work_id":"0fb09554-8ce6-4366-8b78-f426e318fcfd","year":2022},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2210.10760","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:15a2cd4df299e5b30eb97ddcf5dc93b3891216f93b994a9535062715ac627ac7","observation_id":"cf2b9bf9-1871-443b-93d0-c6a495e08fc7","resolution":{"observed_at":"2026-05-19T17:22:41.958533Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-22T21:23:26.501616+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-22T21:23:26.501616+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.15149","last_updated":"2026-04-16T15:30:10Z","snapshot_observed_at":"2026-07-06T23:02:44.990811Z","submitted_at":"2026-04-16T15:30:10Z","title":"LLMs Gaming Verifiers: RLVR can Lead to Reward Hacking","version":1},"cited_work":{"arxiv_id":"2604.15149","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.15149","snapshot_observed_at":"2026-07-04T08:29:41.270827Z","title":"LLMs Gaming Verifiers: RLVR can Lead to Reward Hacking","venue":"cs.LG","work_id":"5aeeb8b5-a071-4399-88c1-9e8869a33e22","year":2026},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2604.15149","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:0d94e4dcae47c2b51b945df802f938429558a00df2c8106ef892d866d12cabed","observation_id":"8e0d6ab5-b1f0-4275-b9e3-bc374b552458","resolution":{"observed_at":"2026-05-19T17:22:41.950802Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning to Reason for Long-Form Story Generation","venue":null,"work_id":"5b68a718-0451-4271-96c5-30019b8a4fa9","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:1c84e2902aaacb7baf521c6e7ffb170cb202ca72bc377afd562412b10953d7bf","observation_id":"b0608133-b245-4ab7-b1bb-08a425e706b2","resolution":{"observed_at":"2026-05-19T17:22:42.175048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13502","last_updated":"2026-08-04T12:28:27Z","snapshot_observed_at":"2026-08-06T14:30:35.562734Z","submitted_at":"2025-06-16T13:58:54Z","title":"BOW: Training Language Models to Reason Over Plausible Next Words","version":3},"cited_work":{"arxiv_id":"2506.13502","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13502","snapshot_observed_at":"2026-08-05T02:57:26.390855Z","title":null,"venue":null,"work_id":"0241028e-59fe-4be2-912f-bd76b6e408c5","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2506.13502","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:b62e9acf4205d0984ffa1c114ecc500bc8e40a2e1fee240c18a8a505d4ecc2f5","observation_id":"3c6377b9-8581-42b0-8f6a-2e27cdac3496","resolution":{"observed_at":"2026-08-05T02:57:26.390855Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Goodman.Learning to Simulate Human Dialogue","venue":null,"work_id":"dbc8f615-5e64-493c-9bc0-f27f3d606af5","year":null},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:b6421e8bb34456e205e72b25c5ffbf40931100b4733dc9dbbb0a466a75d6ee3d","observation_id":"58624433-75e6-4cca-b616-e9c57a00ed2f","resolution":{"observed_at":"2026-05-19T17:22:42.170264Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.04436","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning to simulate human dialogue","venue":null,"work_id":"eaf45b08-651a-4a6c-9159-b1739acb0d4b","year":2026},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:a6f4f0e12831ce6dcc91826196b868135ceea79b0b892a5d8c8e1dcfee092942","observation_id":"cc87e6fa-65c6-4ae4-917b-ff593ee951e2","resolution":{"observed_at":"2026-05-19T17:22:41.974017Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.09629","last_updated":"2024-03-18T07:56:48Z","snapshot_observed_at":"2026-07-31T09:25:34.205362Z","submitted_at":"2024-03-14T17:58:16Z","title":"Quiet-STaR: Language Models Can Teach Themselves to Think Before Speaking","version":2},"cited_work":{"arxiv_id":"2403.09629","doi":"10.48550/arxiv.2403.09629","metadata_source":"pith","pith_arxiv_id":"2403.09629","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Quiet-STaR: Language Models Can Teach Themselves to Think Before Speaking","venue":"cs.CL","work_id":"cd82874a-9e9d-46cb-9716-a06d454a9c7e","year":2024},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2403.09629","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:806ce981aa73890ee929fe3f649f97746ecd14b1dbcb74a6f1c707f437b83adc","observation_id":"faf3e3ba-c466-40e4-940d-44333b642945","resolution":{"observed_at":"2026-05-19T17:22:41.967822Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.08007","last_updated":"2025-06-09T17:59:53Z","snapshot_observed_at":"2026-07-06T21:39:13.304260Z","submitted_at":"2025-06-09T17:59:53Z","title":"Reinforcement Pre-Training","version":1},"cited_work":{"arxiv_id":"2506.08007","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.08007","snapshot_observed_at":"2026-07-03T23:49:02.979112Z","title":"arXiv preprint arXiv:2506.08007 , year=","venue":null,"work_id":"9aea0749-2851-4adc-81d0-8b3c73fd2486","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2506.08007","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:1f8dd44b9f22a4c7701a57f16068326229937715a40dc0f5d712112110408c0b","observation_id":"461999dc-ad6a-4e77-a4ae-6c712e72e1bb","resolution":{"observed_at":"2026-05-19T17:22:41.994710Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.19249","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T02:46:29.025709Z","title":"Reinforcement learning on pre-training data.arXiv preprint arXiv:2509.19249","venue":null,"work_id":"65cab8fa-4ea2-4d2b-b3fa-2e385ed76533","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:018137be6bad6f7b4e2e9605c859bd479979485f44e174f9882c1d32758b8381","observation_id":"14f06e40-d62f-4722-9953-a81ef7d994ad","resolution":{"observed_at":"2026-05-19T17:22:41.980815Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.01265","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T23:49:02.947153Z","title":"RLP: Reinforcement as a Pretraining Objective","venue":null,"work_id":"69d504b4-ce19-4333-b738-34d428fa2a1a","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:a6870ac91ca6cd2bdccbef09cbc9c8430eca3da62fdace6067b8c419678f3141","observation_id":"52d0e59a-c204-41fb-87cb-89abf8f4502d","resolution":{"observed_at":"2026-05-19T17:22:41.947707Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.07127","last_updated":"2025-04-29T03:59:01Z","snapshot_observed_at":"2026-07-06T19:48:32.368124Z","submitted_at":"2024-11-11T16:58:36Z","title":"Benchmarking LLMs' Judgments with No Gold Standard","version":2},"cited_work":{"arxiv_id":"2411.07127","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.07127","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Benchmarking LLMs’ Judgments with No Gold Standard","venue":null,"work_id":"60fa0606-c518-4dc5-b961-5ef303ee08b3","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2411.07127","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:5593cf57e6c823a08e86ecbc34c39488da0d8067d1b8172c390edd9a9b3e9038","observation_id":"4b577c64-f3c9-487b-a38f-4266168dba15","resolution":{"observed_at":"2026-05-19T17:22:41.932179Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:8f2eb4d037dca6ad664d9c4378c0d445346989abe8fec04f3fe487fd8a0d6a41","observation_id":"ed3cf325-6311-46df-86db-1fe326369ea3","resolution":{"observed_at":"2026-05-19T17:22:41.907325Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09766","last_updated":"2024-06-07T09:41:36Z","snapshot_observed_at":"2026-08-04T10:35:44.012115Z","submitted_at":"2023-11-16T10:43:26Z","title":"LLMs as Narcissistic Evaluators: When Ego Inflates Evaluation Scores","version":4},"cited_work":{"arxiv_id":"2311.09766","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.09766","snapshot_observed_at":"2026-07-07T12:53:50.204969Z","title":"Llms as narcissistic evaluators: When ego inflates evaluation scores, 2024 b","venue":"cs.CL","work_id":"78b2f05f-bf08-4c50-83e9-84526d25af2d","year":2023},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2311.09766","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:d7edb120af770bc9177d01e3c3759be709e7523aee0070289e76ebb99b639138","observation_id":"a2461b85-a270-4cc7-8fe6-478af047a7a1","resolution":{"observed_at":"2026-05-19T17:22:41.940048Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13076","last_updated":"2024-04-15T16:49:59Z","snapshot_observed_at":"2026-07-06T18:02:53.240463Z","submitted_at":"2024-04-15T16:49:59Z","title":"LLM Evaluators Recognize and Favor Their Own Generations","version":1},"cited_work":{"arxiv_id":"2404.13076","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.13076","snapshot_observed_at":"2026-07-10T10:27:02.394079Z","title":"LLM Evaluators Recognize and Favor Their Own Generations","venue":"cs.CL","work_id":"ec9ad2bb-47a2-46c7-b8e9-05f69142decd","year":2024},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2404.13076","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:56a751b62064e89fac845601d9c7a974da2c0a2c55effba46af04fe1970e566f","observation_id":"c8c56dd2-ae08-41c1-a880-a984a517a4e2","resolution":{"observed_at":"2026-05-22T18:44:28.909456Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.08794","last_updated":"2026-06-11T04:21:36Z","snapshot_observed_at":"2026-07-06T21:55:50.206625Z","submitted_at":"2025-07-11T17:55:22Z","title":"One Token to Fool LLM-as-a-Judge","version":3},"cited_work":{"arxiv_id":"2507.08794","doi":"10.48550/arxiv.2507.08794","metadata_source":"pith","pith_arxiv_id":"2507.08794","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"One token to fool llm-as-a-judge","venue":"cs.LG","work_id":"f77305eb-5f89-4afb-84f3-51d864f3f3fa","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2507.08794","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:ec039c622cb2007d5350d3343b353fec9cc372359d733c0dc6d66c1a22a154ce","observation_id":"8d4139fe-d65f-4338-a535-7e6fd2ff8b87","resolution":{"observed_at":"2026-06-12T02:08:19.458599Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:590155279aed809f64f60bc7ba7663a383aa08107e9d67aa658df909964fe694","observation_id":"e606b530-0887-4636-bc31-b5cbd5bd3382","resolution":{"observed_at":"2026-05-19T17:22:41.920948Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b5f625bd-c714-4636-9b45-8b13b4c28d09","year":2026},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:f69c84ab662a9fe18ae77044a890320d515bd40ff0841e86bcb02bb786ce5ceb","observation_id":"fa42aa84-eff7-4c62-9ce6-2c2702b6391c","resolution":{"observed_at":"2026-05-19T17:22:42.167950Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20534","last_updated":"2026-02-03T04:57:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-28T05:35:43Z","title":"Kimi K2: Open Agentic Intelligence","version":2},"cited_work":{"arxiv_id":"2507.20534","doi":"10.1145/3448609","metadata_source":"pith","pith_arxiv_id":"2507.20534","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi K2: Open Agentic Intelligence","venue":"cs.LG","work_id":"7f18284c-12d3-4137-bea1-1da97e8cf3c1","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2507.20534","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:e3a128d39c9b29259369b159cb0a8daf91725d36265db01056ca64fdfd9268b1","observation_id":"b02aba94-8097-4662-8b40-85ac683a0cee","resolution":{"observed_at":"2026-05-19T17:22:41.914260Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-25T01:23:16.170083+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T01:23:16.170083+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://deploymentsafety.openai.com/gpt-5-5/gpt-5- 5.pdf","venue":null,"work_id":"d7fa5a5b-e26a-46cf-a6c5-a5edba426168","year":2026},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:fe2e8468280b8b82783c7211fa3e5955395ad19ecc188814b68e0f289160489a","observation_id":"1e19df9f-2a4b-4d80-b51f-ee26f6c7c996","resolution":{"observed_at":"2026-05-19T17:22:42.164288Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://anthropic.com/claude- opus- 4- 7- system-card","venue":null,"work_id":"a87cfe1d-d292-4014-bd9a-4e16ad8d5363","year":2026},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:57ff724ae92714fc59bf356832cbb0547a1fc625f10b6f69f03e41ceb31b3a54","observation_id":"610333d7-3b26-44d5-a96c-a44f2ea0b0c5","resolution":{"observed_at":"2026-05-19T17:22:42.161863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"492fc964-c7ca-4e60-ba16-ab7bd084a030","year":2026},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:f96aced7080267c1fbd3271b5ac6358782a96014fd025801f1c8f18f7de646f7","observation_id":"6d19877f-fbb5-4d58-a92f-ba3ed10dc329","resolution":{"observed_at":"2026-05-19T17:22:42.166164Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":"2408.03314","doi":"10.18653/v1/2025.acl-long.1486","metadata_source":"pith","pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","venue":"cs.LG","work_id":"a8d50b24-bdf5-46ed-bc4f-2927dfd81f1d","year":2024},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:e17016876f18ae89f32ef855450b1498f94923bc311c3d5b8f9d94d9368e9e11","observation_id":"44500651-9eda-4dcb-9ed6-403dcb4beb60","resolution":{"observed_at":"2026-05-19T17:22:41.928675Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1038/s42256-020-00257-z","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Wichmann","venue":"Nature Machine Intelligence","work_id":"d6c3d284-997a-46fa-b978-a5dd28d7658b","year":2020},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:1aef337df33b020d14c584951305e50fb2fdce9cfef33659f9db43f32545ad93","observation_id":"965dae13-ce52-4aa5-8420-e16417529331","resolution":{"observed_at":"2026-05-19T17:22:41.683792Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-12T02:49:23.819858+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T02:49:23.819858+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":"2106.09685","doi":"10.4088/pcc.v03n0609","metadata_source":"pith","pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","venue":"cs.CL","work_id":"0426219a-789e-4964-adc8-a04538510818","year":2021},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:308c2a57b49376b2da8a893f566b216ada6c5203249e8d491ecc97a8092c90f9","observation_id":"ab477633-0f8b-42a5-bf17-d76af8ba216b","resolution":{"observed_at":"2026-05-19T17:22:41.991528Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.06031","last_updated":"2016-06-20T09:37:17Z","snapshot_observed_at":"2026-08-05T16:27:22.031013Z","submitted_at":"2016-06-20T09:37:17Z","title":"The LAMBADA dataset: Word prediction requiring a broad discourse context","version":1},"cited_work":{"arxiv_id":"1606.06031","doi":null,"metadata_source":"pith","pith_arxiv_id":"1606.06031","snapshot_observed_at":"2026-07-10T15:27:20.134743Z","title":"The LAMBADA dataset: Word prediction requiring a broad discourse context","venue":"cs.CL","work_id":"3775e6d1-2f28-4791-82a9-538c0507512c","year":2016},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/1606.06031","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:42e49285610bbe48b51f5ace6e7c628bbcba155d44e9ae40cfa76c6a2d160278","observation_id":"663c3c43-ec16-4717-b034-328e589a3bb1","resolution":{"observed_at":"2026-05-19T17:22:41.688057Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.07830","last_updated":"2019-05-19T23:57:23Z","snapshot_observed_at":"2026-07-31T00:09:56.948833Z","submitted_at":"2019-05-19T23:57:23Z","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","version":1},"cited_work":{"arxiv_id":"1905.07830","doi":"10.48550/arxiv.1905.07830","metadata_source":"pith","pith_arxiv_id":"1905.07830","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","venue":"cs.CL","work_id":"79f44c0c-96f4-4edb-bc50-a3c9d6b85936","year":2019},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/1905.07830","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:54ee3cb172b4afd2706b8354df4cdc01b7096a9147000de4909233315eca582b","observation_id":"8936e89d-57d2-40b9-b95e-fb3cfb7b122f","resolution":{"observed_at":"2026-05-19T17:22:41.680582Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9166.321917","doi":"10.1145/3219166.3219172","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eliciting Expertise without Verification","venue":null,"work_id":"3c3283f9-1228-40c5-8624-f78c720384e9","year":2018},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:a25ab4fca38c16bbd38575f265862b7a7898346cbce6a34549d773688d5934dc","observation_id":"387ffd55-39a0-462d-9f19-0f6b0f8e56ed","resolution":{"observed_at":"2026-05-19T17:22:41.660724Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0865.367353","doi":"10.1145/3670865.3673532","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eliciting Informative Text Evaluations with Large Language Models","venue":null,"work_id":"a6762561-ecd3-443e-8697-faff8213c77b","year":2024},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:771608ffb1fc022ff46c949d844bdc90abc3c33eef692702424da4f75b843d26","observation_id":"db21fb2e-80e3-4708-94cd-29ed59ea279b","resolution":{"observed_at":"2026-05-19T17:22:41.654523Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.07658","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"From raw corpora to domain benchmarks: Automated evaluation of LLM domain expertise","venue":null,"work_id":"011756a9-5365-4094-8b03-2c780bff7524","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:63b31666c1a2ca39ede26bdb311f140c048f3620eeccb8b1a7edab9f3b05f0bd","observation_id":"10970b81-9a0b-4435-a71a-ef1def50328c","resolution":{"observed_at":"2026-05-19T17:22:41.918111Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19618","last_updated":"2025-05-28T14:42:09Z","snapshot_observed_at":"2026-07-06T20:58:23.695292Z","submitted_at":"2025-03-25T13:03:09Z","title":"Beyond Verifiable Rewards: Scaling Reinforcement Learning for Language Models to Unverifiable Data","version":2},"cited_work":{"arxiv_id":"2503.19618","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.19618","snapshot_observed_at":"2026-07-03T20:38:55.982366Z","title":"Beyond Verifiable Rewards: Scaling Reinforcement Learning in Language Models to Unverifiable Data","venue":null,"work_id":"03ad91b0-bdfa-4379-ba78-275537609db4","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2503.19618","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:8221c9eff1d94a5d82b439a41401b0026e07ee599998ab65a9cfef9f0a594cc9","observation_id":"4f3fdcbb-411b-4dd3-b2fa-138136a6315d","resolution":{"observed_at":"2026-05-19T17:22:41.904395Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02179","last_updated":"2023-11-28T17:47:32Z","snapshot_observed_at":"2026-07-06T16:56:46.051958Z","submitted_at":"2023-11-28T17:47:32Z","title":"Training Chain-of-Thought via Latent-Variable Inference","version":1},"cited_work":{"arxiv_id":"2312.02179","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.02179","snapshot_observed_at":"2026-07-04T10:59:45.797413Z","title":"Training Chain-of-Thought via Latent-Variable Inference","venue":null,"work_id":"1dc7e5c9-18d3-4876-b8d7-f3e1919fcb86","year":2023},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2312.02179","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:5ef9dea3bf7470607a5b6c2d0bf537532f1c7931f131b0374ececba8ecddbe58","observation_id":"c380f5f3-a7fd-4121-a9d9-989a848e34bc","resolution":{"observed_at":"2026-05-19T17:22:42.011138Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.04363","last_updated":"2024-03-13T22:48:14Z","snapshot_observed_at":"2026-08-06T06:24:20.124815Z","submitted_at":"2023-10-06T16:36:08Z","title":"Amortizing intractable inference in large language models","version":2},"cited_work":{"arxiv_id":"2310.04363","doi":"10.48550/arxiv.2310.04363","metadata_source":"pith","pith_arxiv_id":"2310.04363","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2310.04363 , year=","venue":"cs.LG","work_id":"82cbff04-d5b7-4cc2-bc8f-ada86d8df9a8","year":2023},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2310.04363","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:69b34769d30cae74aa076c0c565a3a02b6e22a2e3743ae4af0ae3f71a36d0baa","observation_id":"da76e2c7-a966-4617-992f-297c9dbfd6de","resolution":{"observed_at":"2026-05-19T17:22:41.987390Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"NOVER: Incentive Training for Language Models via Verifier-Free Rein- forcement Learning","venue":null,"work_id":"6111f4c5-3be5-4c7a-a5b3-65435a5cd19f","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:e534e547ef7050203db95932336ad9b99ec2bebcde6482cbce7e4b397ac9d873","observation_id":"90816fac-5eb9-4ded-8ebd-bd67a07304c1","resolution":{"observed_at":"2026-05-19T17:22:42.159617Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13351","last_updated":"2026-05-07T20:19:13Z","snapshot_observed_at":"2026-07-31T17:48:52.945760Z","submitted_at":"2025-06-16T10:43:38Z","title":"Direct Reasoning Optimization: Token-Level Reasoning Reflectivity Meets Rubric Gates for Unverifiable Tasks","version":3},"cited_work":{"arxiv_id":"2506.13351","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.13351","snapshot_observed_at":"2026-07-02T23:07:27.269519Z","title":"Direct Reasoning Optimization: Token-Level Reasoning Reflectivity Meets Rubric Gates for Unverifiable Tasks","venue":"cs.CL","work_id":"83c204e8-dd80-4a60-9021-954166f727a0","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2506.13351","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:3bc2c9f8c77726a54050b1802304e3911f7ae9bd5442f820b3b9038464721719","observation_id":"3984fa88-075b-41d5-be28-a20913e35529","resolution":{"observed_at":"2026-05-19T17:22:41.983841Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21493","last_updated":"2025-05-27T17:56:27Z","snapshot_observed_at":"2026-07-06T21:31:35.572708Z","submitted_at":"2025-05-27T17:56:27Z","title":"Reinforcing General Reasoning without Verifiers","version":1},"cited_work":{"arxiv_id":"2505.21493","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.21493","snapshot_observed_at":"2026-07-01T20:56:13.505929Z","title":"Reinforcing general reasoning without verifiers","venue":null,"work_id":"2ec388db-b5cc-4baa-b30c-a65df5ea5bf7","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2505.21493","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:b53459f57c08cb46c96d0408fd15b2c42da6e1af62e3569da103827f75f6fba9","observation_id":"8727f605-db0e-467b-8e17-1e63a94ffd2b","resolution":{"observed_at":"2026-05-19T17:22:41.894869Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.18254","last_updated":"2025-06-23T02:56:36Z","snapshot_observed_at":"2026-07-06T21:46:02.163343Z","submitted_at":"2025-06-23T02:56:36Z","title":"RLPR: Extrapolating RLVR to General Domains without Verifiers","version":1},"cited_work":{"arxiv_id":"2506.18254","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.18254","snapshot_observed_at":"2026-07-04T06:29:38.225670Z","title":"RLPR: Extrapolating RLVR to general domains without verifiers.arXiv preprint arXiv:2506.18254","venue":null,"work_id":"bfa10c9c-ac28-4d18-80d0-26a0bae3986f","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2506.18254","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:b0e9a4ca7c6ff7e393589ddde488e91ea4d69a97027752a508c2472ec5447e07","observation_id":"72cb19f3-c96d-4734-b47f-68a9b6dacd23","resolution":{"observed_at":"2026-05-19T17:22:41.998128Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.03979","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T20:00:08.391833Z","title":"Likelihood- based reward designs for general llm reasoning","venue":null,"work_id":"0f0b3017-6fef-42b2-b7ba-6358bd0f635d","year":2026},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:203181b584d9744a67f6247c887b8b7269db83dfcad2087477c7c946ef799a08","observation_id":"aa6dc1e9-1f3f-4ec2-8fcc-0db2082d1529","resolution":{"observed_at":"2026-05-19T17:22:42.014458Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.20050","last_updated":"2023-05-31T17:24:00Z","snapshot_observed_at":"2026-08-05T13:11:04.104454Z","submitted_at":"2023-05-31T17:24:00Z","title":"Let's Verify Step by Step","version":1},"cited_work":{"arxiv_id":"2305.20050","doi":"10.1007/bf00262952","metadata_source":"pith","pith_arxiv_id":"2305.20050","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Let's Verify Step by Step","venue":"cs.LG","work_id":"6d05b790-04c5-4fd2-91b2-ba1dfdd5770f","year":2023},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2305.20050","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:03f478ece45a9831d708f771fb2e172e75741cec0459fb9e447b0c4ac4dd3d42","observation_id":"9f47c5dd-87c3-4803-a923-2bb0cd99653c","resolution":{"observed_at":"2026-05-19T17:22:41.970558Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15240","last_updated":"2025-02-22T10:21:46Z","snapshot_observed_at":"2026-07-06T19:06:41.697671Z","submitted_at":"2024-08-27T17:57:45Z","title":"Generative Verifiers: Reward Modeling as Next-Token Prediction","version":3},"cited_work":{"arxiv_id":"2408.15240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.15240","snapshot_observed_at":"2026-07-08T21:25:38.629808Z","title":"Generative Verifiers: Reward Modeling as Next-Token Prediction","venue":"cs.LG","work_id":"dbe9c116-8030-4e7c-a259-b52c80846d0e","year":2024},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2408.15240","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:531edf802765a9dfc01fb167ec69b1f3e5eb443abf1732d4e0b8797114c84177","observation_id":"935fe3b7-6a3a-4a89-a513-58cd0581c024","resolution":{"observed_at":"2026-05-19T17:22:41.977768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.13692","last_updated":"2024-08-01T17:18:54Z","snapshot_observed_at":"2026-07-06T18:48:37.723242Z","submitted_at":"2024-07-18T16:58:18Z","title":"Prover-Verifier Games improve legibility of LLM outputs","version":2},"cited_work":{"arxiv_id":"2407.13692","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.13692","snapshot_observed_at":"2026-07-09T02:35:53.846004Z","title":"Prover-verifier games improve legibility of llm outputs","venue":"cs.CL","work_id":"2bebdadc-2df7-4816-93c8-70252a1645eb","year":2024},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2407.13692","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:41005ac340e7c8a94380fbaf449204dbefd62857a2ec4f6a58888ba56fbbd78e","observation_id":"6cbfbf02-1ba1-421c-8ad5-e79267653787","resolution":{"observed_at":"2026-05-19T17:22:41.910605Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.17995","last_updated":"2026-04-14T03:25:43Z","snapshot_observed_at":"2026-08-02T10:58:14.984747Z","submitted_at":"2025-09-22T16:36:56Z","title":"Variation in Verification: Understanding Verification Dynamics in Large Language Models","version":2},"cited_work":{"arxiv_id":"2509.17995","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.17995","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Variation in Verification: Understanding Verification Dynamics in Large Language Models","venue":"cs.CL","work_id":"f614fdf8-06a5-45a4-a7b0-d2ded78a7289","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2509.17995","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:095fdac52b49ba5389ce3219937f4942c634414f56b100490aa6096d5f39cb83","observation_id":"6970f9e5-0ad5-4862-aae5-490007e2211c","resolution":{"observed_at":"2026-05-19T17:22:41.963957Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.06621","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:18:58.357399Z","title":"Reward under attack: Analyzing the robustness and hackability of process reward models","venue":null,"work_id":"98c8e480-4613-4072-b7f8-621056dc4118","year":2026},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:5d71809075cdadfb8e658d491a8bac9410d4c4f63e38d4cc1444a08caf9c50cc","observation_id":"437854ed-ec01-456b-a1ee-d7a6194df3c7","resolution":{"observed_at":"2026-05-19T17:22:41.897933Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02900","last_updated":"2024-11-05T01:44:14Z","snapshot_observed_at":"2026-07-06T18:25:35.827334Z","submitted_at":"2024-06-05T03:41:37Z","title":"Scaling Laws for Reward Model Overoptimization in Direct Alignment Algorithms","version":2},"cited_work":{"arxiv_id":"2406.02900","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.02900","snapshot_observed_at":"2026-07-08T21:25:38.650008Z","title":"B., Finn, C., and Niekum, S","venue":"cs.LG","work_id":"f959e0ed-a9df-409e-a8db-ec3d23671531","year":2024},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"cited_paper":"/paper/2406.02900","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:4ca93064566b20beeff37b09dfa8038de767a7e7a7ac605823e5f50af7be891f","observation_id":"c85fdfea-fd70-4bc8-876d-ca122f1d053c","resolution":{"observed_at":"2026-05-19T17:22:41.901099Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.19248","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T08:09:40.616096Z","title":"Inference-time reward hacking in large language models.arXiv preprint arXiv:2506.19248","venue":null,"work_id":"59a377b6-99b4-4400-827c-9b61c2b62d66","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:39.969447Z"},"links":{"citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:40358d8fa18d70bc4e88158ffaecd2aa7b59d21c28b4c964b0e3d2e58e9c06f3","observation_id":"5ab4af3b-05cb-4a8c-8778-5f03d9347430","resolution":{"observed_at":"2026-05-19T17:22:41.924264Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-01T14:39:13.243144Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities"},"reference_resolution":{"displayed":50,"state_counts":{"malformed_identifier":1,"metadata_mismatch":4,"parse_uncertain":0,"unresolved":2,"verified_exact":37,"verified_fuzzy":6},"total_outbound_references":50},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 50 of 50 outbound references and 0 inbound Pith citation observations for arXiv:2605.10810."}