{"as_of":"2026-08-12T06:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d0d6a0c00c96e4ad0b536fcaed4c2afdffeef09870c7b6b11661a3394cc88b48","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":146,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T21:17:01.988107Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":1084,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"1805.00899","last_updated":"2018-10-22T17:36:07Z","snapshot_observed_at":"2026-08-02T15:33:17.783178Z","submitted_at":"2018-05-02T16:27:32Z","title":"AI safety via debate","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-13T21:22:01.289451Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/1805.00899"},"observation_digest":"sha256:faee5932fbb44eb30bd91e63222462b3d9c637c3a492e0318da30eabb5036839","observation_id":"0eaca5ad-6510-4b06-932c-d60eb3d17d03","resolution":{"observed_at":"2026-05-13T21:22:01.318969Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"1906.09114","last_updated":"2019-07-09T21:47:50Z","snapshot_observed_at":"2026-08-02T23:36:33.424039Z","submitted_at":"2019-06-20T06:32:36Z","title":"Near-optimal Bayesian Solution For Unknown Discrete Markov Decision Process","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-25T20:08:03.014744Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/1906.09114"},"observation_digest":"sha256:c3c63a0a785fab9faec8bf4faa4abdd8d1a47d1b061a6916ccd50d3ff5607daf","observation_id":"7ee7e5c5-693f-44f8-8edc-2d9c24b21975","resolution":{"observed_at":"2026-05-25T20:11:11.531556Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"1906.09627","last_updated":"2019-06-23T19:06:09Z","snapshot_observed_at":"2026-08-05T22:55:46.577982Z","submitted_at":"2019-06-23T19:06:09Z","title":"Inductive general game playing","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-25T17:47:10.838972Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/1906.09627"},"observation_digest":"sha256:4fbc816609031f0f67b0221d58da62daf7c6068f5ccfc759a22fbdcd12233bf6","observation_id":"fc83e076-d3cd-4d7e-9e6d-0e9a58688e09","resolution":{"observed_at":"2026-05-25T17:51:06.211247Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"1906.10124","last_updated":"2019-06-25T15:18:10Z","snapshot_observed_at":"2026-08-11T18:52:35.309042Z","submitted_at":"2019-06-25T15:18:10Z","title":"On Multi-Agent Learning in Team Sports Games","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-25T15:51:28.289206Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/1906.10124"},"observation_digest":"sha256:359410fb51cb6e9468547d66be342ae6a8d2e0f8a0ee67d5b42975a4c4eae0ca","observation_id":"f0a6c56a-c69e-47ea-9569-3676dba578e4","resolution":{"observed_at":"2026-05-25T15:56:01.105162Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"1906.12266","last_updated":"2019-06-28T15:35:11Z","snapshot_observed_at":"2026-07-06T08:03:38.968054Z","submitted_at":"2019-06-28T15:35:11Z","title":"Growing Action Spaces","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-25T13:24:59.682609Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/1906.12266"},"observation_digest":"sha256:5074d249ce16dce4171273ee14cd64ad606bc8487f6b1bb0ef07f141e75ea2a2","observation_id":"5d96b3c8-3011-408c-8fc9-7a61619caf22","resolution":{"observed_at":"2026-05-25T13:25:52.372334Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"1907.06508","last_updated":"2019-07-11T13:02:25Z","snapshot_observed_at":"2026-07-06T08:07:47.943239Z","submitted_at":"2019-07-11T13:02:25Z","title":"General Board Game Playing for Education and Research in Generic AI Game Learning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-24T23:20:39.142086Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/1907.06508"},"observation_digest":"sha256:7fe23370d890fcc1d6767be4e28e95fc62d2cd930ad52324c5e1d4494407e01f","observation_id":"a640cb50-f99a-495d-8840-4f16cc9e329e","resolution":{"observed_at":"2026-05-24T23:25:03.970949Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"1910.07113","last_updated":"2019-10-16T00:59:05Z","snapshot_observed_at":"2026-08-02T15:37:37.200292Z","submitted_at":"2019-10-16T00:59:05Z","title":"Solving Rubik's Cube with a Robot Hand","version":1},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-05-15T09:38:28.621842Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/1910.07113"},"observation_digest":"sha256:5826d13410809a19f0511d58f69395fdd81db1a9cb9243e9bfe1b6e7887507d2","observation_id":"0abe9487-8397-4217-af11-baa2fe004dee","resolution":{"observed_at":"2026-05-15T09:38:28.895897Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"1911.01547","last_updated":"2019-11-25T13:02:04Z","snapshot_observed_at":"2026-07-06T08:34:41.399203Z","submitted_at":"2019-11-05T00:31:38Z","title":"On the Measure of Intelligence","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-12T13:05:34.106110Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/1911.01547"},"observation_digest":"sha256:37833eaa47267e8a951f987a38b8150467da0baf583d2170fbf954e591fd8a07","observation_id":"0c02283a-22cb-4f6a-933e-a72bbff4bd6e","resolution":{"observed_at":"2026-05-12T13:05:34.171922Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2009.03393","last_updated":"2020-09-07T19:50:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-09-07T19:50:10Z","title":"Generative Language Modeling for Automated Theorem Proving","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-23T05:18:10.620262Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2009.03393"},"observation_digest":"sha256:681d9ac8eb8acafdc1eb88e2363f91265320af3b9ae004f4c9bd5741f04e9132","observation_id":"8381a50b-b274-4684-9ff2-d1a1133fc4d7","resolution":{"observed_at":"2026-05-23T05:18:10.793298Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2207.05221","last_updated":"2022-11-21T16:38:35Z","snapshot_observed_at":"2026-08-06T08:34:11.887259Z","submitted_at":"2022-07-11T22:59:39Z","title":"Language Models (Mostly) Know What They Know","version":4},"reference_index":208,"source":"arxiv_source","source_observed_at":"2026-05-10T15:42:47.274448Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2207.05221"},"observation_digest":"sha256:dd50e2e4f549c7b1f685483711e6dc3755dea28cab2bb8d89f9f013742cd51e6","observation_id":"73302cc3-b4e0-40d2-86f1-59127324da1f","resolution":{"observed_at":"2026-05-10T15:42:47.476702Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2304.01373","last_updated":"2023-05-31T17:54:07Z","snapshot_observed_at":"2026-08-07T23:49:29.478548Z","submitted_at":"2023-04-03T20:58:15Z","title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","version":2},"reference_index":171,"source":"arxiv_source","source_observed_at":"2026-05-15T17:45:17.540282Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2304.01373"},"observation_digest":"sha256:a49c3058152545bba73739220df666d0d75928eb0968f45bb0bc05ceefa3e168","observation_id":"5b094429-2807-4abd-bfe9-6fa891c7c3ef","resolution":{"observed_at":"2026-05-15T17:45:17.827643Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2305.14992","last_updated":"2023-10-23T07:24:28Z","snapshot_observed_at":"2026-07-06T15:32:25.931739Z","submitted_at":"2023-05-24T10:28:28Z","title":"Reasoning with Language Model is Planning with World Model","version":2},"reference_index":130,"source":"arxiv_source","source_observed_at":"2026-05-17T01:49:28.796581Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2305.14992"},"observation_digest":"sha256:b94fe1d22461599eb7fd74b59b58f4baf42dee9eb3d4ddf9fdb31c2c8e94cb5e","observation_id":"5d1d5c01-630d-491d-9367-0b0ca6d33ad0","resolution":{"observed_at":"2026-05-17T01:49:28.964959Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2310.02635","last_updated":"2026-04-23T15:06:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-04T07:56:42Z","title":"Reinforcement Learning with Foundation Priors: Let the Embodied Agent Efficiently Learn on Its Own","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-24T06:40:00.328012Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2310.02635"},"observation_digest":"sha256:c85bb0c207511837debddd0ef98e3565553dd776f9f20cc2971787354af81ad8","observation_id":"bc59fdd5-f186-4d79-8d15-01b35dd647b3","resolution":{"observed_at":"2026-05-24T06:44:02.659097Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2310.04406","last_updated":"2024-06-06T02:51:17Z","snapshot_observed_at":"2026-08-06T23:53:10.606737Z","submitted_at":"2023-10-06T17:55:11Z","title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-16T23:26:04.445225Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2310.04406"},"observation_digest":"sha256:66d0e468fe86bb3f6db5576992d244af2c9116b1715c0c2dad85e83cf48e1608","observation_id":"82bf8c01-107d-4a33-bae1-912a5bf63842","resolution":{"observed_at":"2026-05-16T23:26:04.520860Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2310.06114","last_updated":"2024-09-26T17:14:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-09T19:42:22Z","title":"Learning Interactive Real-World Simulators","version":3},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-16T02:15:18.265190Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2310.06114"},"observation_digest":"sha256:58f29f7147b880bfaf524911c092908a178fd4d9e81ba5b7f7d285ccdc23791d","observation_id":"54a06b1a-4101-4aec-89aa-8058b0f629eb","resolution":{"observed_at":"2026-05-16T02:15:18.500659Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-11T21:17:01.988107Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.04847","last_updated":"2024-12-06T08:35:33Z","snapshot_observed_at":"2026-08-11T21:10:54.800850Z","submitted_at":"2024-12-06T08:35:33Z","title":"MTSpark: Enabling Multi-Task Learning with Spiking Neural Networks for Generalist Agents","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T21:17:01.988107Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2412.04847"},"observation_digest":"sha256:2e3d99a0075032bb8c90dbcc5aad7bca9534b75f47701c89105292a16de841d3","observation_id":"93501dce-6bb5-4f92-a496-7ee3a7ad24a9","resolution":{"observed_at":"2026-08-11T21:17:01.988107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-11T21:08:18.413650Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.05196","last_updated":"2025-03-11T17:25:01Z","snapshot_observed_at":"2026-08-11T20:47:15.217043Z","submitted_at":"2024-12-06T17:20:50Z","title":"Exponential Speedups by Rerooting Levin Tree Search","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T21:08:18.413650Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2412.05196"},"observation_digest":"sha256:ee8c20edb835d49a7a2d47012125922c837f63986b3f4cfbde27cb99fe79b131","observation_id":"02838376-9eab-42fe-a73f-7553b9b6589b","resolution":{"observed_at":"2026-08-11T21:08:18.413650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-11T19:51:19.986569Z","title":"arXiv preprint arXiv:1712.01815 (2017) https: //doi.org/10.48550/arXiv.1712.01815","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.06333","last_updated":"2025-05-24T12:49:59Z","snapshot_observed_at":"2026-08-11T19:43:36.756536Z","submitted_at":"2024-12-09T09:34:40Z","title":"Augmenting the action space with conventions to improve multi-agent cooperation in Hanabi","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T19:51:19.986569Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2412.06333"},"observation_digest":"sha256:49c578d1c3bac2e8b99e01366bde900ed5c2164ce73b195aab9279a88b9e0fcf","observation_id":"d6e5d3b2-1623-4529-a6c0-c9278cffbdb6","resolution":{"observed_at":"2026-08-11T19:51:19.986569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-11T19:09:12.875718Z","title":"Lillicrap, Karen Simonyan, and Demis Hassabis","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.07186","last_updated":"2024-12-10T04:52:26Z","snapshot_observed_at":"2026-08-11T19:00:16.329599Z","submitted_at":"2024-12-10T04:52:26Z","title":"Monte Carlo Tree Search based Space Transfer for Black-box Optimization","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T19:09:12.875718Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2412.07186"},"observation_digest":"sha256:4ca24d05ff2c30df68665b3ebe106d3287e83dded22c89744830587f3a68c851","observation_id":"bb8e9ed8-20a7-4907-a60d-8687c5fc7c36","resolution":{"observed_at":"2026-08-11T19:09:12.875718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-11T14:02:49.125617Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.12544","last_updated":"2024-12-28T02:30:02Z","snapshot_observed_at":"2026-08-11T13:56:28.774062Z","submitted_at":"2024-12-17T05:10:21Z","title":"Seed-CTS: Unleashing the Power of Tree Search for Superior Performance in Competitive Coding Tasks","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-11T14:02:49.125617Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2412.12544"},"observation_digest":"sha256:4a324bd17404de655f14deda6fa2d1855f564da6369b7ef3bb73386524033eb1","observation_id":"7dcbdc2d-a295-438d-ad05-1702f5c610a1","resolution":{"observed_at":"2026-08-11T14:02:49.125617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-11T13:42:49.630897Z","title":"Lillicrap, Karen Simonyan, and Demis Hassabis","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.12881","last_updated":"2024-12-17T13:05:36Z","snapshot_observed_at":"2026-08-11T13:36:30.247947Z","submitted_at":"2024-12-17T13:05:36Z","title":"RAG-Star: Enhancing Deliberative Reasoning with Retrieval Augmented Verification and Refinement","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-11T13:42:49.630897Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2412.12881"},"observation_digest":"sha256:7b7a484068e643954711145ee5a6a23b9ab8ad984add94bc1e883691af7ce371","observation_id":"33e22423-7f97-4549-aba4-c0eb5d12a761","resolution":{"observed_at":"2026-08-11T13:42:49.630897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-11T11:08:56.533394Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.15797","last_updated":"2024-12-20T11:14:29Z","snapshot_observed_at":"2026-08-11T11:03:24.099864Z","submitted_at":"2024-12-20T11:14:29Z","title":"Ensembling Large Language Models with Process Reward-Guided Tree Search for Better Complex Reasoning","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-11T11:08:56.533394Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2412.15797"},"observation_digest":"sha256:1138589595a490cb2ab80ee9516cef63fa59d576a370648c75fb20a006441908","observation_id":"5ca98572-cd7b-402e-832b-2556fcb4081e","resolution":{"observed_at":"2026-08-11T11:08:56.533394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-11T05:32:05.125316Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.17397","last_updated":"2024-12-23T08:51:48Z","snapshot_observed_at":"2026-08-11T19:08:40.776110Z","submitted_at":"2024-12-23T08:51:48Z","title":"Towards Intrinsic Self-Correction Enhancement in Monte Carlo Tree Search Boosted Reasoning via Iterative Preference Learning","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-11T05:32:05.125316Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2412.17397"},"observation_digest":"sha256:c6adc711517cce215844f9bfb7da52f092434bd948829dee72903d63c52dd346","observation_id":"fe3f1d92-ba3d-40d6-be20-3812083b914e","resolution":{"observed_at":"2026-08-11T05:32:05.125316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-11T05:11:36.105634Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.17799","last_updated":"2025-05-16T21:19:02Z","snapshot_observed_at":"2026-08-11T05:06:19.564554Z","submitted_at":"2024-12-23T18:57:00Z","title":"Automating the Search for Artificial Life with Foundation Models","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T05:11:36.105634Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2412.17799"},"observation_digest":"sha256:ebef76298bb1dea4f47b54b926443245049da9f6c9586ff16814726c03dcc82e","observation_id":"6a231f84-78a8-49c5-bc27-805f2653b871","resolution":{"observed_at":"2026-08-11T05:11:36.105634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-10T19:52:42.047606Z","title":"Mastering Chess and Shogi by Self - Play with a General Reinforcement Learning Algorithm , December 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.09646","last_updated":"2025-01-16T16:38:33Z","snapshot_observed_at":"2026-08-10T19:45:52.525390Z","submitted_at":"2025-01-16T16:38:33Z","title":"NS-Gym: Open-Source Simulation Environments and Benchmarks for Non-Stationary Markov Decision Processes","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-10T19:52:42.047606Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2501.09646"},"observation_digest":"sha256:159e213d301a2f59a77252fc443e17aa161bea5e11ca53636f8219be2abb47c3","observation_id":"6ba3a3f3-7c77-478a-9912-b58697dd6e95","resolution":{"observed_at":"2026-08-10T19:52:42.047606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-10T19:32:44.230387Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.09959","last_updated":"2025-01-17T05:21:49Z","snapshot_observed_at":"2026-08-10T19:27:34.306445Z","submitted_at":"2025-01-17T05:21:49Z","title":"A Survey on Multi-Turn Interaction Capabilities of Large Language Models","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-10T19:32:44.230387Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2501.09959"},"observation_digest":"sha256:2b9489000a95ad6fa0d5ef6505e2e403c8f7d126ad2d01ce4727ebd14b5d4d19","observation_id":"99a84f7f-9ae3-499d-913b-b7166dcb7dd7","resolution":{"observed_at":"2026-08-10T19:32:44.230387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-10T19:27:46.327698Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.10053","last_updated":"2025-08-27T07:38:28Z","snapshot_observed_at":"2026-08-11T01:13:11.117790Z","submitted_at":"2025-01-17T09:16:13Z","title":"AirRAG: Autonomous Strategic Planning and Reasoning Steer Retrieval Augmented Generation","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-10T19:27:46.327698Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2501.10053"},"observation_digest":"sha256:5835d7e8d5cd78678e9222f6ed1f3247935cc728a83d2b93921724a3d8b5c9fd","observation_id":"92a1232c-ee55-4e65-b270-8433a89c431c","resolution":{"observed_at":"2026-08-10T19:27:46.327698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-11T12:04:30.315475Z","title":"mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.10388","last_updated":"2025-01-23T22:53:04Z","snapshot_observed_at":"2026-08-11T11:58:26.601527Z","submitted_at":"2024-12-19T09:40:40Z","title":"Beyond the Sum: Unlocking AI Agents Potential Through Market Forces","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T12:04:30.315475Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2501.10388"},"observation_digest":"sha256:0cce07dbb16c5e03d6a3985832670074a67e546899a3cf67a6b64f64612eb2c3","observation_id":"c01dde67-806f-4de8-81c3-a6d3d5c47842","resolution":{"observed_at":"2026-08-11T12:04:30.315475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-10T19:53:42.711581Z","title":"Silver, T","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.10476","last_updated":"2025-01-16T17:09:57Z","snapshot_observed_at":"2026-08-12T00:45:38.741906Z","submitted_at":"2025-01-16T17:09:57Z","title":"Revisiting Rogers' Paradox in the Context of Human-AI Interaction","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T19:53:42.711581Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2501.10476"},"observation_digest":"sha256:45784c560a479e3cc1cd61739404889946f0a37ec55e09d9bcd2a761a472899b","observation_id":"eeb9d69a-03ce-4f16-8622-951c2ef8f387","resolution":{"observed_at":"2026-08-10T19:53:42.711581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-10T16:56:50.980566Z","title":"Available: http://arxiv.org/abs/1712.01815","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.12703","last_updated":"2025-07-21T04:43:33Z","snapshot_observed_at":"2026-08-10T16:50:44.883401Z","submitted_at":"2025-01-22T08:18:56Z","title":"HEPPO-GAE: Hardware-Efficient Proximal Policy Optimization with Generalized Advantage Estimation","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-10T16:56:50.980566Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2501.12703"},"observation_digest":"sha256:f7bcd692c9a45812ee7fbf8842839fe41f50a4a9f9999f35f581761bdde948aa","observation_id":"f6ad1393-60cf-4f1d-bc29-dc0df14e6709","resolution":{"observed_at":"2026-08-10T16:56:50.980566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-10T14:48:24.308054Z","title":"https://arxiv.org/abs/1712.01815","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.15001","last_updated":"2025-02-13T04:15:47Z","snapshot_observed_at":"2026-08-10T14:41:13.183280Z","submitted_at":"2025-01-25T00:29:24Z","title":"What if Eye...? Computationally Recreating Vision Evolution","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T14:48:24.308054Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2501.15001"},"observation_digest":"sha256:b967293352e6ec2951a449a3aa9969ce93a13d86551adce4c973974ebf0535db","observation_id":"7d8b1132-0942-45fb-bbf0-f08977538e45","resolution":{"observed_at":"2026-08-10T14:48:24.308054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-09T18:23:05.665783Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.00633","last_updated":"2025-02-02T02:45:20Z","snapshot_observed_at":"2026-08-10T06:28:50.465489Z","submitted_at":"2025-02-02T02:45:20Z","title":"Lipschitz Lifelong Monte Carlo Tree Search for Mastering Non-Stationary Tasks","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-09T18:23:05.665783Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.00633"},"observation_digest":"sha256:cf02d5e25ff4b807b5ea95699afb16ccf0d95907ceb3d9d7a5445f3445ce29a4","observation_id":"e6bc3594-ea50-41aa-b4b8-b1c5686cd307","resolution":{"observed_at":"2026-08-09T18:23:05.665783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-09T17:29:42.716190Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.00915","last_updated":"2025-02-02T21:05:12Z","snapshot_observed_at":"2026-08-09T17:12:48.198896Z","submitted_at":"2025-02-02T21:05:12Z","title":"A Variational Inequality Approach to Independent Learning in Static Mean-Field Games","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-09T17:29:42.716190Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.00915"},"observation_digest":"sha256:61d0c029a7cb032d98e21efabd56acb04b3ba42fa21ab1ded6317ea5a7cf4ff1","observation_id":"8f82252d-cc4f-4fde-8f0b-458d0bec97b8","resolution":{"observed_at":"2026-08-09T17:29:42.716190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-09T15:09:20.573488Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.01492","last_updated":"2025-02-03T16:26:17Z","snapshot_observed_at":"2026-08-10T08:00:48.918214Z","submitted_at":"2025-02-03T16:26:17Z","title":"Develop AI Agents for System Engineering in Factorio","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-09T15:09:20.573488Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.01492"},"observation_digest":"sha256:3e2d9f873034937597e58b0b0921a633315c1fb3a0405a3862919ea2e999b2d6","observation_id":"71f9a99c-4ded-44b2-bf3f-f543e487623b","resolution":{"observed_at":"2026-08-09T15:09:20.573488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-09T13:35:05.249257Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.02060","last_updated":"2025-02-04T07:13:21Z","snapshot_observed_at":"2026-08-11T00:15:11.314057Z","submitted_at":"2025-02-04T07:13:21Z","title":"CH-MARL: Constrained Hierarchical Multiagent Reinforcement Learning for Sustainable Maritime Logistics","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-09T13:35:05.249257Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.02060"},"observation_digest":"sha256:3b1719252bac0d25be4cbe767592af0e776a626659533cd82abe0291fe289a78","observation_id":"42f2bac0-71e6-4c31-847b-d3c222489a9a","resolution":{"observed_at":"2026-08-09T13:35:05.249257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-09T13:17:17.462579Z","title":"Mastering Chess and Shogi by Self-Play with a Gen- eral Reinforcement Learning Algorithm,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.02133","last_updated":"2025-02-04T09:06:07Z","snapshot_observed_at":"2026-08-11T19:30:29.660874Z","submitted_at":"2025-02-04T09:06:07Z","title":"Synthesis of Model Predictive Control and Reinforcement Learning: Survey and Classification","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-09T13:17:17.462579Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.02133"},"observation_digest":"sha256:d94ccd8dc304ccd578707f15706109461043a3a5d080268c3203bb1771dea8bb","observation_id":"b2531ecd-f9c3-43ff-b4dd-243733311256","resolution":{"observed_at":"2026-08-09T13:17:17.462579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2502.03387","last_updated":"2025-07-29T16:23:02Z","snapshot_observed_at":"2026-08-08T04:13:22.884923Z","submitted_at":"2025-02-05T17:23:45Z","title":"LIMO: Less is More for Reasoning","version":3},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-05-17T02:11:36.932541Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.03387"},"observation_digest":"sha256:1499b353721517bb52dc655697132e0899db42a2e72b322d33ccb6d43c2f24d5","observation_id":"b4b80ebf-5a1e-400e-8a2b-dc6c33c4488a","resolution":{"observed_at":"2026-05-17T02:11:37.593984Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-09T00:33:49.251920Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.04402","last_updated":"2025-02-06T08:07:35Z","snapshot_observed_at":"2026-08-09T16:49:25.983822Z","submitted_at":"2025-02-06T08:07:35Z","title":"Beyond Interpolation: Extrapolative Reasoning with Reinforcement Learning and Graph Neural Networks","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-09T00:33:49.251920Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.04402"},"observation_digest":"sha256:f9b66846553bedb442dfe84df18738093ae21de06f8dbad2dd86e7d22208c5c4","observation_id":"e3f38b74-f831-47e1-a2a6-91f7c19c4fca","resolution":{"observed_at":"2026-08-09T00:33:49.251920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-08T21:40:26.896146Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.04751","last_updated":"2025-02-07T08:36:39Z","snapshot_observed_at":"2026-08-09T03:57:00.818497Z","submitted_at":"2025-02-07T08:36:39Z","title":"Holistically Guided Monte Carlo Tree Search for Intricate Information Seeking","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-08T21:40:26.896146Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.04751"},"observation_digest":"sha256:15e6c33974ccc2a76d21e7adfd4689e9a1d6775181e7b6b57784e64bf81c468e","observation_id":"13770f35-3892-417f-b92e-e4aff51cea6e","resolution":{"observed_at":"2026-08-08T21:40:26.896146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-08T20:51:10.107804Z","title":"Mastering chess and shogi by self-play with a general reinforce- ment learning algorithm","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.05244","last_updated":"2025-02-07T14:29:07Z","snapshot_observed_at":"2026-08-09T14:58:49.748389Z","submitted_at":"2025-02-07T14:29:07Z","title":"Probabilistic Artificial Intelligence","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T20:51:10.107804Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.05244"},"observation_digest":"sha256:b91ae8a8efe2cc57c4d768830c178a9d3c0715e00cad16c5e10e5046f10fb70b","observation_id":"a36ca27a-c381-46e4-a85b-a0960fc6b2fa","resolution":{"observed_at":"2026-08-08T20:51:10.107804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-08T20:20:35.334579Z","title":"Silver, T","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.05253","last_updated":"2025-02-07T17:21:16Z","snapshot_observed_at":"2026-08-11T17:14:04.961970Z","submitted_at":"2025-02-07T17:21:16Z","title":"LLMs Can Teach Themselves to Better Predict the Future","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-08T20:20:35.334579Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.05253"},"observation_digest":"sha256:57d4880723b899770bfec9641e9889c92819f893bc59c03c67e1f821ea8cbb9d","observation_id":"2ac05b3c-09a7-4cc2-8855-cce582f6580b","resolution":{"observed_at":"2026-08-08T20:20:35.334579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-08T19:53:25.780992Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.05314","last_updated":"2025-02-13T23:45:02Z","snapshot_observed_at":"2026-08-09T11:53:14.188856Z","submitted_at":"2025-02-07T20:28:44Z","title":"Two-Player Zero-Sum Differential Games with One-Sided Information","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-08T19:53:25.780992Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.05314"},"observation_digest":"sha256:f39afd574c4024cab3e845fef4ff0269fb7fd78ed0eb130e2c938c997a1ecd16","observation_id":"8ffa25c5-17da-454a-801f-0ae3455dcae6","resolution":{"observed_at":"2026-08-08T19:53:25.780992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-09T11:20:31.870961Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.06813","last_updated":"2025-02-04T22:08:20Z","snapshot_observed_at":"2026-08-09T12:19:32.223241Z","submitted_at":"2025-02-04T22:08:20Z","title":"Policy Guided Tree Search for Enhanced LLM Reasoning","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-09T11:20:31.870961Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.06813"},"observation_digest":"sha256:b4ef707fcf80ca6af9eeefeab44aa91ccad4eefea62fcd3629a5005bad8e9ad0","observation_id":"51fffe79-a280-413d-b851-57e036a9be1b","resolution":{"observed_at":"2026-08-09T11:20:31.870961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-08T12:17:46.058042Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.07586","last_updated":"2025-02-11T14:34:05Z","snapshot_observed_at":"2026-08-10T04:58:17.549351Z","submitted_at":"2025-02-11T14:34:05Z","title":"We Can't Understand AI Using our Existing Vocabulary","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-08T12:17:46.058042Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2502.07586"},"observation_digest":"sha256:e3b62cb0f9efe2e57581ea4a2088335099339709bf099254adb9659c23b43cf5","observation_id":"058bfed7-c6c2-4d82-ac31-7778c2ce6739","resolution":{"observed_at":"2026-08-08T12:17:46.058042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2503.01804","last_updated":"2026-04-09T14:35:28Z","snapshot_observed_at":"2026-07-06T20:45:55.263034Z","submitted_at":"2025-03-03T18:33:46Z","title":"$\\texttt{SEM-CTRL}$: Semantically Controlled Decoding","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2503.01804"},"observation_digest":"sha256:6ddf24d3a66c47adbc4e8925dca47dc32610dbc3ab661638cb7f5ab9ee4d4814","observation_id":"1d75d4da-4c03-46e7-a36e-48077c96761a","resolution":{"observed_at":"2026-05-23T01:32:21.677175Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2503.13377","last_updated":"2025-06-29T08:11:35Z","snapshot_observed_at":"2026-08-02T21:01:10.455745Z","submitted_at":"2025-03-17T17:04:20Z","title":"Time-R1: Post-Training Large Vision Language Model for Temporal Video Grounding","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-17T02:40:06.454859Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2503.13377"},"observation_digest":"sha256:6450b4c02c05e9df5166a037861020d191576d62033ece2af2175fda6c974f2b","observation_id":"bec547c2-e15e-4994-9ade-b935f7dcc8b1","resolution":{"observed_at":"2026-05-17T02:40:06.528100Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2504.13541","last_updated":"2026-04-17T06:31:21Z","snapshot_observed_at":"2026-07-06T21:11:24.852957Z","submitted_at":"2025-04-18T08:12:59Z","title":"Scalable Multi-Task Learning through Spiking Neural Networks with Adaptive Task-Switching Policy for Intelligent Autonomous Agents","version":5},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-22T19:45:07.172201Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2504.13541"},"observation_digest":"sha256:a19546d8ae5518a052dced161498353ed24d1b357b143810c6d99b0c66701a13","observation_id":"9d420f29-727e-4b67-8ded-c93a1f5a20f8","resolution":{"observed_at":"2026-05-22T19:47:01.203155Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T15:42:23.541553Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.14107","last_updated":"2025-05-29T08:24:00Z","snapshot_observed_at":"2026-08-07T23:55:00.741921Z","submitted_at":"2025-05-20T09:14:53Z","title":"DiagnosisArena: Benchmarking Diagnostic Reasoning for Large Language Models","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:42:23.541553Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2505.14107"},"observation_digest":"sha256:fcce14684bc4df193ce5ed40497ab4cb38551df56d826e28902e21b9256025bc","observation_id":"143ce6d6-de68-44ed-84b8-5b90c3a4a49a","resolution":{"observed_at":"2026-08-07T15:42:23.541553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T14:12:05.275950Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.19683","last_updated":"2025-05-26T08:44:53Z","snapshot_observed_at":"2026-08-08T09:25:43.689145Z","submitted_at":"2025-05-26T08:44:53Z","title":"Large Language Models for Planning: A Comprehensive and Systematic Survey","version":1},"reference_index":221,"source":"pdf_text","source_observed_at":"2026-08-07T14:12:05.275950Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2505.19683"},"observation_digest":"sha256:8cef56d32fa2a862cf9739f6610adfc07b48832f856d3c03b4d4df50e6d2133a","observation_id":"fb50a7b3-b697-4853-8960-64934c712f57","resolution":{"observed_at":"2026-08-07T14:12:05.275950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T13:33:42.027095Z","title":"Tapley, A., Gatesman, K., Robaina, L., Bissey, B., and Weissman, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.21414","last_updated":"2025-05-27T16:41:23Z","snapshot_observed_at":"2026-08-11T01:45:10.079536Z","submitted_at":"2025-05-27T16:41:23Z","title":"A Framework for Adversarial Analysis of Decision Support Systems Prior to Deployment","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T13:33:42.027095Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2505.21414"},"observation_digest":"sha256:3f1ee8e9a7d7d0b607ad26621ebedefc4d6593cce1f185c3942bd6f1a015aacf","observation_id":"ce2edbaa-5f35-45f9-8e1e-649dd05b8711","resolution":{"observed_at":"2026-08-07T13:33:42.027095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T13:05:28.844213Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-11T16:38:12.898922Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.844213Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:8b07b9489be01af44c4ec2a7bd14106ea25af801f6fdba566c79b07ecd859b83","observation_id":"f9703d2b-f199-4dd6-8544-44b9b91e86e5","resolution":{"observed_at":"2026-08-07T13:05:28.844213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T12:45:28.836184Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.24034","last_updated":"2025-06-02T01:49:51Z","snapshot_observed_at":"2026-08-07T23:13:48.148542Z","submitted_at":"2025-05-29T22:14:15Z","title":"LlamaRL: A Distributed Asynchronous Reinforcement Learning Framework for Efficient Large-scale LLM Training","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-07T12:45:28.836184Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2505.24034"},"observation_digest":"sha256:65be147f93383c2585e7a1a90d290d3b6d921f9f83eba379346437cfe8e74900","observation_id":"bd313a9f-5794-4dbb-9f22-8b1be12cc3de","resolution":{"observed_at":"2026-08-07T12:45:28.836184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T11:35:02.041298Z","title":"Silver, T","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.02205","last_updated":"2025-06-30T20:54:54Z","snapshot_observed_at":"2026-08-10T15:19:49.201433Z","submitted_at":"2025-06-02T19:44:40Z","title":"Bregman Centroid Guided Cross-Entropy Method","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:35:02.041298Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2506.02205"},"observation_digest":"sha256:b3d6ec285672fcb52ca55b8f3a95e4aee362b63bce3daae92885e190daa4236f","observation_id":"9c343496-eb31-43df-98bc-4424873161c5","resolution":{"observed_at":"2026-08-07T11:35:02.041298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T11:31:02.529238Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.02355","last_updated":"2025-06-20T04:14:47Z","snapshot_observed_at":"2026-08-08T15:13:20.527982Z","submitted_at":"2025-06-03T01:15:15Z","title":"Rewarding the Unlikely: Lifting GRPO Beyond Distribution Sharpening","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T11:31:02.529238Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2506.02355"},"observation_digest":"sha256:f1cd9a8d24631a73ffa46257cf11e96ebb1dd468c45b11450a464e0cb2d9fdf3","observation_id":"419ef1f1-a177-49e1-8a10-5c862632ae18","resolution":{"observed_at":"2026-08-07T11:31:02.529238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T10:36:35.772152Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.04821","last_updated":"2025-06-05T09:40:47Z","snapshot_observed_at":"2026-08-10T20:15:44.637146Z","submitted_at":"2025-06-05T09:40:47Z","title":"LogicPuzzleRL: Cultivating Robust Mathematical Reasoning in LLMs via Reinforcement Learning","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T10:36:35.772152Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2506.04821"},"observation_digest":"sha256:f5e7efa8aa8b019ddcdc315212ec1a15b2ed24e0cea9a62d2e9e33f9f944beb5","observation_id":"eb642055-069f-49a9-a58c-65798539348c","resolution":{"observed_at":"2026-08-07T10:36:35.772152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T10:37:34.688532Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.04892","last_updated":"2025-06-05T11:19:26Z","snapshot_observed_at":"2026-08-09T09:40:05.828074Z","submitted_at":"2025-06-05T11:19:26Z","title":"Learning to Plan via Supervised Contrastive Learning and Strategic Interpolation: A Chess Case Study","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T10:37:34.688532Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2506.04892"},"observation_digest":"sha256:6f631cd2d9f700f34b0d6c6d8bf5ed255fe68d4c8647c0e35545c49a345b52d9","observation_id":"96b5d1a6-42ac-479d-9522-b113d999f338","resolution":{"observed_at":"2026-08-07T10:37:34.688532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T05:51:30.719001Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm.arXiv preprint arXiv:1712.01815,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.06923","last_updated":"2025-06-07T21:23:00Z","snapshot_observed_at":"2026-08-12T02:01:39.057364Z","submitted_at":"2025-06-07T21:23:00Z","title":"Boosting LLM Reasoning via Spontaneous Self-Correction","version":1},"reference_index":2008,"source":"pdf_text","source_observed_at":"2026-08-07T05:51:30.719001Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2506.06923"},"observation_digest":"sha256:c0492545ae9b45b06f672d6729216b9a847e5cdd2b9d4d03a733e27897918cfc","observation_id":"1c712f5b-949f-4280-a6ee-3b068b91deb3","resolution":{"observed_at":"2026-08-07T05:51:30.719001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T01:08:24.323361Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.11902","last_updated":"2025-06-13T15:52:37Z","snapshot_observed_at":"2026-08-07T00:58:51.821987Z","submitted_at":"2025-06-13T15:52:37Z","title":"TreeRL: LLM Reinforcement Learning with On-Policy Tree Search","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T01:08:24.323361Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2506.11902"},"observation_digest":"sha256:6a7f56bbfd8fc25a39c663bb8f746e2e608789025f6946e43c47b7efa566cd25","observation_id":"c8f2eee2-1bb1-44aa-bead-6ad9de2b6459","resolution":{"observed_at":"2026-08-07T01:08:24.323361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-06T23:48:43.148148Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.16352","last_updated":"2025-06-19T14:29:48Z","snapshot_observed_at":"2026-08-06T23:41:32.615263Z","submitted_at":"2025-06-19T14:29:48Z","title":"Data-Driven Policy Mapping for Safe RL-based Energy Management Systems","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T23:48:43.148148Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2506.16352"},"observation_digest":"sha256:7095b550d3d13a0184c8127c23795807ae6ac4873a8b522530b0e94943a051d6","observation_id":"99368f62-42ff-4941-b9c3-b2d8ea26e4b2","resolution":{"observed_at":"2026-08-06T23:48:43.148148Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-06T22:49:39.387074Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm.arXiv preprint arXiv:1712.01815,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.387074Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:c34f50b0ba36295fc30be19c6cc24a71163730bd6d9a86239db4cac1773b941f","observation_id":"dd2ea97c-aa52-4054-999b-1fa2e8176f5e","resolution":{"observed_at":"2026-08-06T22:49:39.387074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T05:07:39.615289Z","title":"Mastering Chess and Shogi by self-play with a general reinforcement learning algorithm,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-10T12:28:57.343429Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.615289Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:764fa51d4c58a24d96a17c8415d2591ae7dc28f053f7223f415aaf2fb7c88ee9","observation_id":"3d9dc32f-db41-44a7-8c75-c83962b8fee6","resolution":{"observed_at":"2026-08-07T05:07:39.615289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-06T21:26:50.406662Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.00257","last_updated":"2025-06-30T20:47:50Z","snapshot_observed_at":"2026-08-06T21:18:08.304320Z","submitted_at":"2025-06-30T20:47:50Z","title":"Gym4ReaL: A Suite for Benchmarking Real-World Reinforcement Learning","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-06T21:26:50.406662Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2507.00257"},"observation_digest":"sha256:cf89be454377a617398601dfc36feb839af02cc793b36548ca5b4217e9acae3f","observation_id":"d5e27596-3d6c-4b64-9aa1-59f9f7e7627a","resolution":{"observed_at":"2026-08-06T21:26:50.406662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-06T21:13:15.204837Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.00726","last_updated":"2025-08-27T22:56:13Z","snapshot_observed_at":"2026-08-09T03:39:48.813789Z","submitted_at":"2025-07-01T13:16:34Z","title":"Can Large Language Models Develop Strategic Reasoning? Post-training Insights from Learning Chess","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T21:13:15.204837Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2507.00726"},"observation_digest":"sha256:4868baacd6e82ececd0a0c506d4fb2510698d4392438d0438af50829c6e5bac5","observation_id":"9e9e4522-fbce-4ae5-ac6d-afa829209472","resolution":{"observed_at":"2026-08-06T21:13:15.204837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-06T21:31:51.321684Z","title":"Mastering Chess and Shogi by Self - Play with a General Reinforcement Learning Algorithm , December 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.02969","last_updated":"2025-06-30T15:06:17Z","snapshot_observed_at":"2026-08-10T04:51:54.795421Z","submitted_at":"2025-06-30T15:06:17Z","title":"Reinforcement Learning for Automated Cybersecurity Penetration Testing","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T21:31:51.321684Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2507.02969"},"observation_digest":"sha256:6e20a383ad3a3862a41eac12ea78078a25a9b1e592faf4eeb203e2889b6b98b6","observation_id":"deed1063-266c-4cc0-a404-41faaf89ca6f","resolution":{"observed_at":"2026-08-06T21:31:51.321684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-06T20:18:25.796266Z","title":"Sifre, Dharshan Kumaran, Thore Graepel, Timothy P","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.03314","last_updated":"2025-07-04T05:54:27Z","snapshot_observed_at":"2026-08-06T20:20:16.141801Z","submitted_at":"2025-07-04T05:54:27Z","title":"Partial Label Learning for Automated Theorem Proving","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T20:18:25.796266Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2507.03314"},"observation_digest":"sha256:147fd9c64c6bbeb2f75ee9f390bd5ab5b7148b022c095102513db2fc6f661db6","observation_id":"edcbd904-06a2-410c-809f-67344d3ac2e1","resolution":{"observed_at":"2026-08-06T20:18:25.796266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-06T19:42:27.951210Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm.arXiv preprint arXiv:1712.01815, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.05169","last_updated":"2026-06-17T19:20:50Z","snapshot_observed_at":"2026-08-11T20:10:01.350597Z","submitted_at":"2025-07-07T16:23:46Z","title":"Critique of World Model","version":5},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T19:42:27.951210Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2507.05169"},"observation_digest":"sha256:155dafa4e052602c6bce9eb76794e725385bf93c29f2272a88ad1c51d36c9209","observation_id":"f7ffe987-480b-4136-8cc5-1641a34a07b1","resolution":{"observed_at":"2026-08-06T19:42:27.951210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-06T18:58:36.167097Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.06825","last_updated":"2025-07-10T09:28:09Z","snapshot_observed_at":"2026-08-08T05:15:59.689599Z","submitted_at":"2025-07-09T13:15:05Z","title":"Artificial Generals Intelligence: Mastering Generals.io with Reinforcement Learning","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T18:58:36.167097Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2507.06825"},"observation_digest":"sha256:cda6bc1423a4920122c7265d81266a1e16b4fad5f8671fe841d0f527e81eb5cb","observation_id":"22e431cc-f1da-48bd-92ef-a9c77a23ed9a","resolution":{"observed_at":"2026-08-06T18:58:36.167097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-06T16:34:25.194886Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm.arXiv preprint arXiv:1712.01815,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.13158","last_updated":"2025-07-17T14:22:24Z","snapshot_observed_at":"2026-08-09T20:19:53.961381Z","submitted_at":"2025-07-17T14:22:24Z","title":"Inverse Reinforcement Learning Meets Large Language Model Post-Training: Basics, Advances, and Opportunities","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-06T16:34:25.194886Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2507.13158"},"observation_digest":"sha256:6ef8d6b86c81ed25317b19a96f5b1423c7da9b16f70e05f5d45f8379fdd0efa4","observation_id":"c36e4d87-d47b-44e8-8317-feb967e0fa88","resolution":{"observed_at":"2026-08-06T16:34:25.194886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-06T12:44:46.361511Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21513","last_updated":"2025-07-29T05:30:57Z","snapshot_observed_at":"2026-08-10T19:52:31.904732Z","submitted_at":"2025-07-29T05:30:57Z","title":"What Does it Mean for a Neural Network to Learn a \"World Model\"?","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T12:44:46.361511Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2507.21513"},"observation_digest":"sha256:ede3a4a0fa0a4ee36ee8abf4db66bcbf5a652e2d3616978c3c0c2fb5792d2e2e","observation_id":"d31759d9-4456-4e96-82a5-6807f12a6c34","resolution":{"observed_at":"2026-08-06T12:44:46.361511Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2507.23773","last_updated":"2026-05-21T08:18:41Z","snapshot_observed_at":"2026-07-06T22:05:57.485783Z","submitted_at":"2025-07-31T17:57:20Z","title":"General Agentic Planning Through Simulative Reasoning with World Models","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-22T12:37:10.956233Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2507.23773"},"observation_digest":"sha256:55dc370d78d380601ecccbbd8de2d53c9428285917394dbe680c0ffe80d7e0f9","observation_id":"774b86a0-9dd7-4312-9234-e4225be897b5","resolution":{"observed_at":"2026-05-22T12:41:33.535741Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T23:22:29.446275Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.05441","last_updated":"2025-08-07T14:31:22Z","snapshot_observed_at":"2026-08-09T07:07:06.006850Z","submitted_at":"2025-08-07T14:31:22Z","title":"Tail-Risk-Safe Monte Carlo Tree Search under PAC-Level Guarantees","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-05T23:22:29.446275Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2508.05441"},"observation_digest":"sha256:6b372259dead5530e6518144bce2333f2e63ff5f1a4a5d0b177a2a6977dfc23e","observation_id":"f6ecdc91-0157-44c1-a375-e8a38252c7e3","resolution":{"observed_at":"2026-08-05T23:22:29.446275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T23:06:27.677046Z","title":", Hubert , T","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.05908","last_updated":"2025-08-08T00:13:12Z","snapshot_observed_at":"2026-08-07T13:27:26.847150Z","submitted_at":"2025-08-08T00:13:12Z","title":"Hybrid Physics-Machine Learning Models for Quantitative Electron Diffraction Refinements","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T23:06:27.677046Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2508.05908"},"observation_digest":"sha256:da648d320bf1c66c702e6c07b6213bb6baf9487f19714a7abcec7bb50d8209d3","observation_id":"562758e3-95f4-4845-be47-56ffaeeb86ab","resolution":{"observed_at":"2026-08-05T23:06:27.677046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T22:45:20.775630Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.06443","last_updated":"2025-08-08T16:36:16Z","snapshot_observed_at":"2026-08-09T05:04:25.875616Z","submitted_at":"2025-08-08T16:36:16Z","title":"The Fair Game: Auditing & Debiasing AI Algorithms Over Time","version":1},"reference_index":117,"source":"arxiv_source","source_observed_at":"2026-08-05T22:45:20.775630Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2508.06443"},"observation_digest":"sha256:1f5838eeaa8b1403f2809c9d338b83f9088cd19b3d036d329e546b578c72ef1c","observation_id":"504cf053-eea8-41b3-8015-1c90d66a0ba9","resolution":{"observed_at":"2026-08-05T22:45:20.775630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T22:06:19.698245Z","title":"P.; Simonyan, K.; and Hassabis, D","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.07522","last_updated":"2025-08-11T00:53:52Z","snapshot_observed_at":"2026-08-10T09:00:44.500380Z","submitted_at":"2025-08-11T00:53:52Z","title":"Evolutionary Optimization of Deep Learning Agents for Sparrow Mahjong","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T22:06:19.698245Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2508.07522"},"observation_digest":"sha256:c04ec8b9f77371eac157f1f015fbc98e6b30481d74bf949b285b5c2454d3eaaa","observation_id":"783b6d4e-df78-4688-b664-bf61f0902eba","resolution":{"observed_at":"2026-08-05T22:06:19.698245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T21:02:04.453521Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.09561","last_updated":"2025-08-13T07:29:40Z","snapshot_observed_at":"2026-08-09T02:42:53.914860Z","submitted_at":"2025-08-13T07:29:40Z","title":"Edge General Intelligence Through World Models and Agentic AI: Fundamentals, Solutions, and Challenges","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-05T21:02:04.453521Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2508.09561"},"observation_digest":"sha256:a36b7535067c646e61ab25f5239d1e24e566e0c729172b1aa2dfd9e12b54ae13","observation_id":"14e5daec-6c9c-4667-8f5b-90fb8fe6d078","resolution":{"observed_at":"2026-08-05T21:02:04.453521Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T14:35:51.350639Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm, 2017a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.21188","last_updated":"2025-09-02T16:27:24Z","snapshot_observed_at":"2026-08-09T18:15:22.003522Z","submitted_at":"2025-08-28T20:02:10Z","title":"Mirage or Method? How Model-Task Alignment Induces Divergent RL Conclusions","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T14:35:51.350639Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2508.21188"},"observation_digest":"sha256:6d867c575352142bb9779c5e6b3d5c89bb95a7ab54b1b1a97cbd3116824fcffb","observation_id":"cde19674-1179-4713-ad61-86d99b27d526","resolution":{"observed_at":"2026-08-05T14:35:51.350639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2509.00338","last_updated":"2026-05-08T15:17:41Z","snapshot_observed_at":"2026-08-02T17:36:08.274992Z","submitted_at":"2025-08-30T03:42:10Z","title":"Scalable Option Learning in High-Throughput Environments","version":3},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-18T20:04:58.064472Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2509.00338"},"observation_digest":"sha256:a516e2fe021ff60ccae0c38f9e989b9ffded2d1b0ac3175a23ed801a260a31c5","observation_id":"b65bf5b5-13a2-49c3-8c8f-d92500571d16","resolution":{"observed_at":"2026-05-18T20:06:49.728038Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-04T16:58:12.987074Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.11233","last_updated":"2025-09-14T12:20:38Z","snapshot_observed_at":"2026-08-10T00:41:12.405586Z","submitted_at":"2025-09-14T12:20:38Z","title":"TransZero: Parallel Tree Expansion in MuZero using Transformer Networks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T16:58:12.987074Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2509.11233"},"observation_digest":"sha256:aba5ddb6deed1e9a20e84a975d83579fc169ec881b06d9fc2a24cfb28928e372","observation_id":"e3d5166e-e192-4b50-9afd-3bbb249ff6f5","resolution":{"observed_at":"2026-08-04T16:58:12.987074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-04T17:57:52.477057Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.14257","last_updated":"2026-07-23T12:32:53Z","snapshot_observed_at":"2026-08-05T11:19:48.609021Z","submitted_at":"2025-09-12T15:34:07Z","title":"Student-Centered Distillation Narrows the Agentic Gap Between Small and Large LLMs","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-04T17:57:52.477057Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2509.14257"},"observation_digest":"sha256:c02c87d6ae275323152d3f98a4610544561690f3751e0368f18d6420bc592109","observation_id":"525fa1de-ba25-4a75-9403-e531e4b1fd88","resolution":{"observed_at":"2026-08-04T17:57:52.477057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-02T16:31:11.111154Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:ac117dd3116ab8b4fb49f0950d8e2d2515d0668fad3e38f04c74f17980753a2b","observation_id":"72b104ca-c66f-4952-b4b4-a77ff7925207","resolution":{"observed_at":"2026-05-21T21:34:22.300517Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-04T10:14:17.110418Z","title":"Silver, T","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2510.11503","last_updated":"2026-07-12T23:02:41Z","snapshot_observed_at":"2026-08-12T00:41:49.610223Z","submitted_at":"2025-10-13T15:12:08Z","title":"People use fast and flat simulation to reason about new games","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-04T10:14:17.110418Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2510.11503"},"observation_digest":"sha256:853794be54d3e46b5080db94bd6abc28c17dc0654ea3fae45a32605bc70af189","observation_id":"353cde8c-65a4-435e-b4be-b5577287cd77","resolution":{"observed_at":"2026-08-04T10:14:17.110418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2511.08717","last_updated":"2026-05-04T18:26:22Z","snapshot_observed_at":"2026-08-11T13:20:33.403540Z","submitted_at":"2025-11-11T19:27:14Z","title":"Optimal control of the future via prospective learning with control","version":4},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-17T23:04:54.956095Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2511.08717"},"observation_digest":"sha256:79433beeddcb5268d1822a0ba955e26743cb9ff8d0969281a6a71945a3c07e80","observation_id":"7e257f57-fbcd-40c5-839b-f9aeb7887c89","resolution":{"observed_at":"2026-05-17T23:05:25.326088Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2512.13961","last_updated":"2026-04-14T15:12:44Z","snapshot_observed_at":"2026-08-07T08:18:31.274999Z","submitted_at":"2025-12-15T23:41:48Z","title":"Olmo 3","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2512.13961"},"observation_digest":"sha256:b2679f76baff7e47a9097b9b0c89fbd8c36599ff8040de09be1a9d53ae1ba458","observation_id":"7410afe3-f7c1-4cd2-b1c9-81ad21f0280e","resolution":{"observed_at":"2026-05-16T21:31:16.815834Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-08-07T05:45:14.900335Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:e695448ca94b4c1ad16fed6452261e7c8581a82b66863ee0b5595a7405e04d56","observation_id":"1a2b69ba-8b4f-4e69-b893-e00b80a8a1c1","resolution":{"observed_at":"2026-05-16T21:11:16.802600Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2512.18552","last_updated":"2026-06-02T06:06:43Z","snapshot_observed_at":"2026-08-03T15:02:08.953642Z","submitted_at":"2025-12-21T00:49:40Z","title":"Toward Training Superintelligent Software Agents through Self-Play SWE-RL","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-21T16:07:48.570995Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2512.18552"},"observation_digest":"sha256:c1e625481e578aadc13b19116903dbe6984b524ebf7858e466a651db4f9ca521","observation_id":"9cc4fbe1-2280-4386-8fd8-d51d67c38b86","resolution":{"observed_at":"2026-05-21T16:10:20.288129Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-03T15:02:13.565152Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.18552","last_updated":"2026-06-02T06:06:43Z","snapshot_observed_at":"2026-08-03T15:02:08.953642Z","submitted_at":"2025-12-21T00:49:40Z","title":"Toward Training Superintelligent Software Agents through Self-Play SWE-RL","version":3},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-03T15:02:13.565152Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2512.18552"},"observation_digest":"sha256:bf8027e1e8ba705de494f143b6fc4c863e0dd0f2e45df3f8006c1b3bd9772e02","observation_id":"718bc116-399b-4cb8-8ebc-c01cdc475e14","resolution":{"observed_at":"2026-08-03T15:02:13.565152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-03T14:24:24.762257Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2512.20806","last_updated":"2026-05-31T13:11:43Z","snapshot_observed_at":"2026-08-06T07:38:24.259000Z","submitted_at":"2025-12-23T22:13:14Z","title":"Safety Alignment of LMs via Non-cooperative Games","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-03T14:24:24.762257Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2512.20806"},"observation_digest":"sha256:3f0ab6e8ffab0f753bed3d7ea3b15f95f719a25cd9392d73e03077d6778689f2","observation_id":"2e6e7293-f947-406f-81aa-c91cd60daeb6","resolution":{"observed_at":"2026-08-03T14:24:24.762257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-03T10:04:42.730729Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2601.11809","last_updated":"2026-07-25T21:30:19Z","snapshot_observed_at":"2026-08-08T05:00:42.535903Z","submitted_at":"2026-01-16T22:22:05Z","title":"Multi-agent DRL-based Lane Change Decision Model for Cooperative Platooning in Mixed Traffic","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-03T10:04:42.730729Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2601.11809"},"observation_digest":"sha256:4334eb34343afcca98915f226df5fa73ddcb63327428aec3bdc6b1cc94db4c46","observation_id":"c278cb97-5e22-497d-89f4-23db7ec1a477","resolution":{"observed_at":"2026-08-03T10:04:42.730729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-03T07:08:33.035932Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2601.21169","last_updated":"2026-01-29T02:11:43Z","snapshot_observed_at":"2026-08-08T13:22:13.997719Z","submitted_at":"2026-01-29T02:11:43Z","title":"Output-Space Search: Targeting LLM Generations in a Frozen Encoder-Defined Output Space","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-03T07:08:33.035932Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2601.21169"},"observation_digest":"sha256:9fc07793bac2eaa88b4cfbce244fc2dfa693e2c8b951d7c0c195a1c36374503e","observation_id":"1fe87430-de22-42e4-abfb-0a787823317f","resolution":{"observed_at":"2026-08-03T07:08:33.035932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-03T05:14:20.763891Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2602.02979","last_updated":"2026-05-24T21:03:36Z","snapshot_observed_at":"2026-08-08T03:39:37.418634Z","submitted_at":"2026-02-03T01:38:53Z","title":"CPMobius: Iterative Coach-Player Reasoning for Data-Free Reinforcement Learning","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-03T05:14:20.763891Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2602.02979"},"observation_digest":"sha256:3171a3717f08e5314739ceb6654678262261646e72527086fbd6c1684ec9d4db","observation_id":"32557d67-3451-44be-8156-77395016f9c0","resolution":{"observed_at":"2026-08-03T05:14:20.763891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2602.08167","last_updated":"2026-05-16T05:42:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-09T00:10:17Z","title":"Self-Supervised Bootstrapping of Action-Predictive Embodied Reasoning","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-21T14:03:48.795572Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2602.08167"},"observation_digest":"sha256:7e06860fb531db292f386f207d148c108c8065ea9883483c05e8427923eb8f76","observation_id":"3e56cbe2-06ef-401e-ac4c-1ef5b24d998a","resolution":{"observed_at":"2026-05-21T14:04:11.930844Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-02T20:42:07.940055Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.22810","last_updated":"2026-06-23T02:26:25Z","snapshot_observed_at":"2026-08-06T15:15:16.654587Z","submitted_at":"2026-02-26T09:50:15Z","title":"Multi-agent imitation learning with function approximation: Linear Markov games and beyond","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-02T20:42:07.940055Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2602.22810"},"observation_digest":"sha256:718f8d43de0e344e63a6dcfb4283e6d1ae989c4c2412716cba7dc5760cde3a23","observation_id":"349ee6f0-86de-41a1-8875-d019a4a94633","resolution":{"observed_at":"2026-08-02T20:42:07.940055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-07-14T23:44:17.042117Z","title":"Silver, T","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2603.10289","last_updated":"2026-06-03T02:01:12Z","snapshot_observed_at":"2026-08-06T19:46:18.775467Z","submitted_at":"2026-03-11T00:15:56Z","title":"Quantum entanglement provides a competitive advantage in adversarial games","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-14T23:44:17.042117Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2603.10289"},"observation_digest":"sha256:3181e5857d6b7152f0be430e865a6c7c6d0997a503d55b80374958809ab3db41","observation_id":"06ebd88d-ee9c-40f6-a8db-7f3cc2982c46","resolution":{"observed_at":"2026-07-14T23:44:17.042117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2604.03312","last_updated":"2026-03-31T22:00:47Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T22:00:47Z","title":"Computer Architecture's AlphaZero Moment: Automated Discovery in an Encircled World","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-08T02:19:33.633470Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2604.03312"},"observation_digest":"sha256:0bbd9601ae753607848fbf8f5e58210360ddfc0491d63c25fd4be5e1a0d0ea4e","observation_id":"f58c790d-9f7d-4d1b-acad-e30429f8a847","resolution":{"observed_at":"2026-05-11T22:51:22.944814Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2604.06228","last_updated":"2026-03-29T21:24:26Z","snapshot_observed_at":"2026-07-06T22:54:44.656835Z","submitted_at":"2026-03-29T21:24:26Z","title":"Probabilistic Language Tries: A Unified Framework for Compression, Decision Policies, and Execution Reuse","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-14T21:21:47.095517Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2604.06228"},"observation_digest":"sha256:796e3595461b5cf47969864c42df1b88ef49400b41377faaa05dfec2fe3fd696","observation_id":"0c3e1e3d-00f1-477e-bd1b-c30f092d37ea","resolution":{"observed_at":"2026-05-14T21:22:58.916568Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2604.09035","last_updated":"2026-04-10T06:53:25Z","snapshot_observed_at":"2026-07-06T22:57:59.555972Z","submitted_at":"2026-04-10T06:53:25Z","title":"Advantage-Guided Diffusion for Model-Based Reinforcement Learning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T17:21:03.813720Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2604.09035"},"observation_digest":"sha256:e3e954ea39a6da6de40f484cab5cc974cf51e004d0113f3fb78bfedee23352a4","observation_id":"7e16cd7d-a8a1-487e-91e0-5dd59846ac29","resolution":{"observed_at":"2026-05-11T07:01:00.636827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2604.10449","last_updated":"2026-04-12T04:15:31Z","snapshot_observed_at":"2026-08-11T00:32:22.016000Z","submitted_at":"2026-04-12T04:15:31Z","title":"AdverMCTS: Combating Pseudo-Correctness in Code Generation via Adversarial Monte Carlo Tree Search","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-10T16:35:16.056397Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2604.10449"},"observation_digest":"sha256:6619ff32e430729cecf10ed1249f73f05baa63159091d767684e4251884a1f70","observation_id":"0b6162e5-857b-4a59-a128-e27154cf945a","resolution":{"observed_at":"2026-05-11T08:40:57.429344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2604.13812","last_updated":"2026-04-15T12:46:40Z","snapshot_observed_at":"2026-08-02T09:56:15.338791Z","submitted_at":"2026-04-15T12:46:40Z","title":"AlphaCNOT: Learning CNOT Minimization with Model-Based Planning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-10T13:10:05.119477Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2604.13812"},"observation_digest":"sha256:ee29b9eabb967871375f46e870560b04fbcb4b06448a85cccf4b745256c9d1a6","observation_id":"2324b110-0069-4883-ba49-ca862400a268","resolution":{"observed_at":"2026-05-10T13:10:25.759222Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2604.15585","last_updated":"2026-04-16T23:37:01Z","snapshot_observed_at":"2026-08-11T13:04:49.837875Z","submitted_at":"2026-04-16T23:37:01Z","title":"PAWN: Piece Value Analysis with Neural Networks","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T10:50:44.608469Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2604.15585"},"observation_digest":"sha256:252dbc34ddef772381fc72e3c7fc449c8a29198b87f0ad4a29c5b117f26cb1aa","observation_id":"5d0a9622-d16a-48a0-bd6e-9fc7119fd27b","resolution":{"observed_at":"2026-05-10T10:55:04.113495Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2604.17952","last_updated":"2026-04-20T08:32:06Z","snapshot_observed_at":"2026-08-11T12:47:59.422079Z","submitted_at":"2026-04-20T08:32:06Z","title":"Causal inference for social network formation","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-05-10T04:08:05.770405Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2604.17952"},"observation_digest":"sha256:16f0c748abe9a43781b0af950741eb3f8ed3e2aa9b6e39d83b005a53154e2be0","observation_id":"632febf0-5ba1-4a6b-ac89-72ce804b4ea6","resolution":{"observed_at":"2026-05-11T12:11:03.925868Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/1712.01815/citation-record","integrity":"/paper/1712.01815/integrity","json":"/paper/1712.01815/citation-record.json","paper":"/paper/1712.01815"},"outbound":[],"paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 100 inbound Pith citation observations for arXiv:1712.01815."}