{"as_of":"2026-08-04T13:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d4277f74a995ebeeac12c96742a4aacebe2ec3a29124e1e6bb33e9d39b73ac70","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-25T01:34:29.047426Z","state":"measured"},{"denominator":51,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":51,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/1907.04214/citation-record","integrity":"/paper/1907.04214/integrity","json":"/paper/1907.04214/citation-record.json","paper":"/paper/1907.04214"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming ; John Wiley & Sons: Hoboken, NJ, USA","venue":null,"work_id":"f41c9f95-1e6a-4c42-9931-20fb747ec2a7","year":1994},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:30c8cb06c900c998fc306e698d7d86b2dc8c9935285f759cad58d42ff204b35d","observation_id":"4ca5afae-4c7a-4482-9413-685f2684cfb7","resolution":{"observed_at":"2026-05-25T01:35:11.442857Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reinforcement Learning: An Introduction ; MIT Press: Cambridge, MA, USA","venue":null,"work_id":"d15465f6-6f08-4374-9554-fe6b8a85c6a1","year":1998},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:20b9ba6016b353d23110f9aaf2082d2b505e51b60f050924279d007e5e9778ce","observation_id":"f80f228d-e4c3-4244-8757-fe889684089d","resolution":{"observed_at":"2026-05-25T01:35:11.446335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A survey on policy search for robotics.Found","venue":null,"work_id":"d8ebce8c-5996-4ec0-9755-1e82be86beb7","year":2013},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:84de4ce18356b4b47b0968bc5c440807f3d2fa279f6bd81659aca784a496bbb2","observation_id":"4d2aabc3-2dd7-47ea-a754-3ed7c198781e","resolution":{"observed_at":"2026-05-25T01:35:11.525896Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dynamic Programming","venue":null,"work_id":"260ed669-89b7-4d70-bf9a-5f36384b2280","year":1957},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:ed3c1d4fbedf761399b410ff21336772b04c44c60156953ace10eec910719ca5","observation_id":"cde12b75-023d-469f-ae36-31326970a7d2","resolution":{"observed_at":"2026-05-25T01:35:11.567568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A Natural Policy Gradient","venue":null,"work_id":"3464596d-6f4b-4981-beb1-10e0ae1c9195","year":2001},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:3e480c99bd085c156d3d65f2c3a7cbd39e4425dde1a6402b4279ecb0a06fa021","observation_id":"c11b2ef4-e891-4128-b155-467c1601c9fa","resolution":{"observed_at":"2026-05-25T01:35:11.564176Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Relative Entropy Policy Search","venue":null,"work_id":"09a98335-442f-41a9-824e-41b4dcceff7b","year":2010},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:40344936c81cdba673555814203c5e6bc53e92c01de4c78aff79088d123c7585","observation_id":"90d87262-2064-4d47-b2d0-ba4a8c1d6f92","resolution":{"observed_at":"2026-05-25T01:35:11.425028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Trust Region Policy Optimization","venue":null,"work_id":"857eae4b-87a9-4f96-ac69-ad6b63f406fd","year":2015},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:b5a07df049d095ce3943fc08002cf6293811b8112be559879463863b32d392e0","observation_id":"a43a6983-4833-4b9f-a93f-96e8b01054ac","resolution":{"observed_at":"2026-05-25T01:35:11.581387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:346078b5af120898d71a41f31d4cb3abe89dec1861fa9b75300070be0cb1d414","observation_id":"4a6c47a1-f714-4e1e-973f-306249e10795","resolution":{"observed_at":"2026-05-25T01:35:10.837484Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improving predictive inference under covariate shift by weighting the log-likelihood function","venue":null,"work_id":"6850400a-d539-417e-bc93-45b043d82209","year":2000},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:ff331d07d26c868af447ac627da6c9469b05bbc7d6b86f98783b0f6ae1c550a1","observation_id":"2415fe7d-8436-4e4c-809e-aa671659114c","resolution":{"observed_at":"2026-05-25T01:35:11.554100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1705.07798","last_updated":"2017-05-22T15:06:25Z","snapshot_observed_at":"2026-07-06T05:43:40.624942Z","submitted_at":"2017-05-22T15:06:25Z","title":"A unified view of entropy-regularized Markov decision processes","version":1},"cited_work":{"arxiv_id":"1705.07798","doi":null,"metadata_source":"pith","pith_arxiv_id":"1705.07798","snapshot_observed_at":"2026-07-09T22:56:37.739931Z","title":"A unified view of entropy-regularized Markov decision processes","venue":"cs.LG","work_id":"80b2d2e4-f964-4515-9237-cb45fb9435e3","year":2017},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"cited_paper":"/paper/1705.07798","citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:28597edecd446b4de5081abbd548277c381cbe6f37935e3f8a8e9cdca9814e38","observation_id":"2ea71631-65a5-44c2-9d08-e1de542bd9de","resolution":{"observed_at":"2026-05-25T01:35:10.850379Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proximal Algorithms","venue":null,"work_id":"c6a964bb-31e6-4570-9fe2-4527a997cd0c","year":2014},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:af916597a68d3381082c85c134f1d2641f0f2aac9e3e082ebd4c2c284d0b9506","observation_id":"fb63fada-3a65-4d37-90eb-fa44b54fd1dc","resolution":{"observed_at":"2026-05-25T01:35:11.588654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1808.08271","last_updated":"2020-09-06T09:54:34Z","snapshot_observed_at":"2026-08-04T11:59:32.793815Z","submitted_at":"2018-08-17T02:33:55Z","title":"An elementary introduction to information geometry","version":2},"cited_work":{"arxiv_id":"1808.08271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1808.08271","snapshot_observed_at":"2026-07-03T12:58:07.577725Z","title":"An elementary introduction to information geometry","venue":null,"work_id":"8310eb4d-7234-4907-8b31-5854e95e7a8f","year":2018},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"cited_paper":"/paper/1808.08271","citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:33cf9b9598f816560f9ae58d1a2019d5cc9de6084c611192404c6fc64b881c0b","observation_id":"a1ad0364-035e-409a-8421-f7723474f946","resolution":{"observed_at":"2026-05-25T01:35:10.807274Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Generative Adversarial Nets","venue":null,"work_id":"f414737f-1ad8-4646-a818-1fd2b6377fd0","year":2014},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:fad86db757986743351318284b35886339924e63a22baf82fcd188b5e13a5e29","observation_id":"dd44e36a-ffbb-43ee-ba07-31510f19f4dc","resolution":{"observed_at":"2026-05-25T01:35:11.421812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Geometrical Insights for Implicit Generative Modeling","venue":null,"work_id":"2953cf14-f3c3-46c3-b9eb-468777df81ed","year":2018},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:fa062f40f570178a6998781f5dde6375d54bc49f79252510571db1ff27e96e1c","observation_id":"05e5a00d-5032-4d40-af0d-fbb6869267ff","resolution":{"observed_at":"2026-05-25T01:35:11.428681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"f-GAN: Training Generative Neural Samplers using Variational Divergence Minimization","venue":null,"work_id":"15848a12-d827-475f-beb6-ca462e79d899","year":2016},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:4027c927fa97560eb7d867aad555c18619157231b40ae77fdadc52db20eaedb3","observation_id":"487d6822-2a58-40dc-8042-d806d3031818","resolution":{"observed_at":"2026-05-25T01:35:11.432302Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Entropic Proximal Mappings with Applications to Nonlinear Programming","venue":null,"work_id":"bb38064f-a17e-4643-919d-d454746cca14","year":1992},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:e87d585edc014491a9660fe86bbaa26302619c4507ef630bba00c041a15cb17e","observation_id":"789a4559-43d4-4aad-a648-1eff5bfd2873","resolution":{"observed_at":"2026-05-25T01:35:11.578067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Problem complexity and method efﬁciency in optimization","venue":null,"work_id":"04b982c7-c478-4a2f-96e6-4bf5e6db34ba","year":1984},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:949fbd00c189ab45d936a2bae22e00a2d99d692d3b4c67340986ca719ad43f66","observation_id":"70394ce9-076b-44bd-833e-f342010eff36","resolution":{"observed_at":"2026-05-25T01:35:11.529250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mirror descent and nonlinear projected subgradient methods for convex optimization","venue":null,"work_id":"65f68408-2176-4a1f-99f4-a106c154fd74","year":2003},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:81190748ce3356ee7f20fbc861fb84c6430e0e83b9048c574a656ea6ad346ec6","observation_id":"4f52db6c-710c-40e0-a4e8-6939914a366a","resolution":{"observed_at":"2026-05-25T01:35:11.560795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A measure of asymptotic efﬁciency for tests of a hypothesis based on the sum of observations","venue":null,"work_id":"904d42a6-a57c-486d-8874-494a72399332","year":1952},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:439679ba4d6b8ce8798ba327a0576753d706fe662a5044786dd5900efbddd626","observation_id":"5b185220-9c80-4744-a835-a5fa4ad98645","resolution":{"observed_at":"2026-05-25T01:35:11.585086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Differential-Geometrical Methods in Statistics ; Springer: New York, NY, USA","venue":null,"work_id":"cf5f994f-1d60-43ee-a66e-b21c2f9970fe","year":1985},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:efad144ea47d858cfe7405bc234d36ce7b7ed02d69ad905f6eaf24c6f34ed405","observation_id":"dd4ad908-5d19-482c-b24a-1c2342b73d21","resolution":{"observed_at":"2026-05-25T01:35:11.453225Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Families of alpha- beta- and gamma- divergences: Flexible and robust measures of Similarities","venue":null,"work_id":"a17c7886-19c8-4420-aa4a-18845ef5aee8","year":2010},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:5974b284566190ca1c988247dbdd29d26f933a40ed28707e37b7f09942046be1","observation_id":"5e885707-4d5f-440b-a1c2-18b4e4388a74","resolution":{"observed_at":"2026-05-25T01:35:11.439500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1512.09075","last_updated":"2016-09-08T14:30:43Z","snapshot_observed_at":"2026-08-04T00:11:34.815681Z","submitted_at":"2015-12-30T19:34:01Z","title":"A Notation for Markov Decision Processes","version":2},"cited_work":{"arxiv_id":"1512.09075","doi":null,"metadata_source":"pith","pith_arxiv_id":"1512.09075","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A Notation for Markov Decision Processes","venue":"cs.AI","work_id":"83526556-76f1-4ad2-ab41-501fa7cac3f4","year":2015},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"cited_paper":"/paper/1512.09075","citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:0cf3d337c525a0aa2a14499064685b04d56d8d1ebcb02787606ad2069a842fc8","observation_id":"a51c0b3d-fe93-40f5-b8d9-a86cb70b0863","resolution":{"observed_at":"2026-05-25T01:35:10.824053Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Policy Gradient Methods for Reinforcement Learning with Function Approximation","venue":null,"work_id":"29ac21ed-9e80-48f5-975a-6207e8b61d16","year":1999},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:d5482cf97359979f4045642c7cdac39cf1b688d5ffe5b5bd015f7dfa49779301","observation_id":"0a7e1787-d63e-43ee-8215-ca5915e9604d","resolution":{"observed_at":"2026-05-25T01:35:11.405133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Natural Actor-Critic","venue":null,"work_id":"a1c643cb-f2d3-482e-9d22-21e794b23ec5","year":2008},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:649c37cbf7613ec5a0fcf652d31a89e656d11204dc1308be0f3cd7164787ae9f","observation_id":"bf933ea6-ea26-4203-84e2-9092d5de6572","resolution":{"observed_at":"2026-05-25T01:35:11.436061Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1506.02438","last_updated":"2018-10-20T18:55:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2015-06-08T11:12:48Z","title":"High-Dimensional Continuous Control Using Generalized Advantage Estimation","version":6},"cited_work":{"arxiv_id":"1506.02438","doi":"10.48550/arxiv.1506.02438","metadata_source":"pith","pith_arxiv_id":"1506.02438","snapshot_observed_at":"2026-07-11T03:57:46.887146Z","title":"High-Dimensional Continuous Control Using Generalized Advantage Estimation","venue":"cs.LG","work_id":"38e3ca94-96f0-4b19-a355-0754931af8be","year":2015},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"cited_paper":"/paper/1506.02438","citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:f213b8fc585814661c4dd07008c1c85cb22d44a3931bd870a5a4da9c067151b3","observation_id":"79d45a85-9698-46b2-9445-717448228549","resolution":{"observed_at":"2026-05-25T01:35:10.856824Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Eine informationstheoretische Ungleichung und ihre Anwendung auf den Beweis der Ergodizität von Markoffschen Ketten","venue":null,"work_id":"3724fcdb-f9b1-47fb-aed4-8baeff677624","year":1963},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:adaf832ecf627cce03fe7f55a66017d34dd9cc72541aa50cc889324b2f8058f8","observation_id":"c37fcee8-ee75-41f7-b769-18a4bef77f5a","resolution":{"observed_at":"2026-05-25T01:35:11.408522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Information Geometric Measurements of Generalisation; Technical Report; Aston University: Birmingham, UK","venue":null,"work_id":"d039e8cc-9d5c-4f74-afae-376652d518c6","year":1995},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:4e48e70e62affbc26d3523b26607b3f410a65c0ea043ba891d970e483e79c30c","observation_id":"7fc63687-8142-4f72-b0f8-176dd898db82","resolution":{"observed_at":"2026-05-25T01:35:11.571324Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Simple statistical gradient-following methods for connectionist reinforcement learning","venue":null,"work_id":"1168409c-6e4f-436d-a40d-0816267bfc73","year":1992},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:642e3cdf70e30e6b0098f0e8ee2c35673fe61d8c5e8a1df2729c4011239b15ef","observation_id":"93b4aca0-472d-4c91-abf1-3f4959107eb7","resolution":{"observed_at":"2026-05-25T01:35:11.418515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Graphical Models, Exponential Families, and Variational Inference","venue":null,"work_id":"9ef564f4-9102-4cea-aa03-02dc80c3ee4d","year":2007},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:adf36640a610b3d23e95094de40806d964a5fc13f3e18fd1ab2ab4c3c4559abf","observation_id":"c49d21f0-fe76-4b3d-a6cf-55bb496aa490","resolution":{"observed_at":"2026-05-25T01:35:11.536527Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Residual Algorithms: Reinforcement Learning with Function Approximation","venue":null,"work_id":"ce4c58ac-9cbc-4ed0-b715-95963e4f2b90","year":1995},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:962d43e3365b0211089e8bf036a99c5dee0c569eeb40e15ac10dd015f28334d7","observation_id":"e944a79c-e81d-42ca-8eaf-90096feb2ec8","resolution":{"observed_at":"2026-05-25T01:35:11.533042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Policy Evaluation with Temporal Differences: A Survey and Comparison","venue":null,"work_id":"efbcc674-6347-4199-bbd4-4897d9b165c3","year":2014},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:748f1bbe2d98bbeaf943cec63230c9cd285f0d60cf76598da0adbdce7447b2f7","observation_id":"5ec633d3-528a-497c-8906-d8410ceae7b6","resolution":{"observed_at":"2026-05-25T01:35:11.449695Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"F-divergence inequalities","venue":null,"work_id":"5dd701f6-7095-454d-a34e-6601e9466994","year":2016},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:d7970691390ce7d3fc1e3a21a9f0b70e0b0503a7eb598d668e98c5fbd26544be","observation_id":"fbc52316-854a-4cb2-a675-63c86c9b421f","resolution":{"observed_at":"2026-05-25T01:35:11.539574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Regret Analysis of Stochastic and Nonstochastic Multi-armed Bandit Problems","venue":null,"work_id":"06923696-4d55-4df7-8c9d-b9d448c8e0ab","year":2012},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:42ac3fd7419bc5fdc7cf51238aacd20c1bded69e53a586b6edcca7274b649d7c","observation_id":"11c4d318-20ed-4983-9986-727bce10d3f1","resolution":{"observed_at":"2026-05-25T01:35:11.543222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The Non-Stochastic Multi-Armed Bandit Problem.SIAM J","venue":null,"work_id":"cd245334-bef0-407b-9103-e5b0be3be32e","year":2003},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:cb43415e9a8e0850c6950d10849847856de2eef4bca452a5f82c83b2b0fecd26","observation_id":"c85c889a-5cb7-49f5-a53c-cf130dccc233","resolution":{"observed_at":"2026-05-25T01:35:11.550566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bayesian Reinforcement Learning: A Survey","venue":null,"work_id":"d2cf6403-f5cf-4852-894d-905f22b6d8bc","year":2015},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:c604778394dc4b1f034cbc6f27633c6eb9b419e94cb58f32dbb53f383c43d774","observation_id":"efc81a9f-2ef9-4f00-a143-ab818269fa93","resolution":{"observed_at":"2026-05-25T01:35:11.557486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.01540","last_updated":"2016-06-05T17:54:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2016-06-05T17:54:48Z","title":"OpenAI Gym","version":1},"cited_work":{"arxiv_id":"1606.01540","doi":"10.1109/jssc.2019","metadata_source":"pith","pith_arxiv_id":"1606.01540","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"OpenAI Gym","venue":"cs.LG","work_id":"6af98f3f-f074-41ae-a689-7dd7b4b8efde","year":2016},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"cited_paper":"/paper/1606.01540","citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:f79aa78a2ef1eebe9d710025abc2ace224d91f37b8514fc9ca9e960d76a12b46","observation_id":"0b1bbcc7-ceaa-4cae-a424-f05b07ec5cfa","resolution":{"observed_at":"2026-05-25T01:35:10.813015Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Information theory of decisions and actions","venue":null,"work_id":"865bd2c9-6595-4982-8db3-7706cf7663e1","year":2011},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:55dbfadf984ed6893bcf95dba0b08592f93deff00f7e871a1d61dc6f2af230f8","observation_id":"686dde6f-72c2-444a-abdd-c72c5f1cad12","resolution":{"observed_at":"2026-05-25T01:35:11.517777Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Autonomy: An information theoretic perspective","venue":null,"work_id":"7e1af8c1-98cc-4b87-95bd-e230b099bd3c","year":2008},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:c10671fc2c92afb76eb0469468049f9949e2122ecb52f584c6f89ac9af00674e","observation_id":"87b5f4df-6a56-4447-859f-7659e7c44911","resolution":{"observed_at":"2026-05-25T01:35:11.521369Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"An information-theoretic approach to curiosity-driven reinforcement learning","venue":null,"work_id":"04d16176-22a6-48f3-ad7b-22400c2b583a","year":2012},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:10f565ab17062cadc9cfea14587510cfe0b8646b43ccb1ae3eb7023171cb1ce3","observation_id":"3eb0224f-ce67-4363-86f7-4a4665a68885","resolution":{"observed_at":"2026-05-25T01:35:11.456346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bounded rationality, abstraction, and hierarchical decision-making: An information-theoretic optimality principle","venue":null,"work_id":"15eea018-8d8a-4a49-8ef4-45a7bc1f7375","year":2015},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:50a70a88884582d5812b448d9f18fc5afd873b04b232dd4a34b8b474bf163211","observation_id":"8aecab8a-e7b4-4bb9-8d89-9c5ce5924af5","resolution":{"observed_at":"2026-05-25T01:35:11.546982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Information theory—the bridge connecting bounded rational game theory and statistical physics","venue":null,"work_id":"ad546958-4746-44c9-9a31-c5f78b44e4f5","year":2006},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:9f01b6946dbaba81b75ce86818f0c7f074e3cce2420c094740de7bbe1c6f7001","observation_id":"78027efc-413b-43e6-bac2-fc4e42585850","resolution":{"observed_at":"2026-05-25T01:35:11.514038Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1901.11275","last_updated":"2019-06-04T07:44:24Z","snapshot_observed_at":"2026-07-06T07:30:14.333834Z","submitted_at":"2019-01-31T09:10:08Z","title":"A Theory of Regularized Markov Decision Processes","version":2},"cited_work":{"arxiv_id":"1901.11275","doi":null,"metadata_source":"pith","pith_arxiv_id":"1901.11275","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A Theory of Regularized Markov Decision Processes","venue":"cs.LG","work_id":"63ce1ecb-43bf-4a4b-a8cd-9073325a100b","year":2019},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"cited_paper":"/paper/1901.11275","citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:dd837fb9d205f24219ff55593aac85a03563c196c546f16b8e4f6b06370fd808","observation_id":"259a522f-702a-4322-9bb2-186974b4928a","resolution":{"observed_at":"2026-05-25T01:35:10.863131Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1903.00725","last_updated":"2019-10-20T15:28:06Z","snapshot_observed_at":"2026-08-03T15:03:08.873912Z","submitted_at":"2019-03-02T15:34:25Z","title":"A Regularized Approach to Sparse Optimal Policy in Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"1903.00725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1903.00725","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A Uniﬁed Framework for Regularized Reinforcement Learning","venue":null,"work_id":"58eb2842-9c04-4ab8-bf23-5bd115b8d789","year":2019},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"cited_paper":"/paper/1903.00725","citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:81fb4daf7f7697ff27d85f98e3f9725ae13185b32334f8dcc909db028e12985b","observation_id":"f189041b-e566-41ff-a85b-262ebde19c7c","resolution":{"observed_at":"2026-05-25T01:35:10.818545Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.03501","last_updated":"2018-02-10T01:57:49Z","snapshot_observed_at":"2026-07-06T06:22:40.962594Z","submitted_at":"2018-02-10T01:57:49Z","title":"Path Consistency Learning in Tsallis Entropy Regularized MDPs","version":1},"cited_work":{"arxiv_id":"1802.03501","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.03501","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Path Consistency Learning in Tsallis Entropy Regularized MDPs","venue":"cs.AI","work_id":"5104169c-c51a-4c0d-990f-904d70c752cd","year":2018},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"cited_paper":"/paper/1802.03501","citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:f3f0ccd52873f14845062ca0700e79a3d9baffa9bcc4bf549b293ebf35582ef0","observation_id":"9af662a9-8bd8-4ba1-bf4f-71415aa378b3","resolution":{"observed_at":"2026-05-25T01:35:10.831638Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1902.00137","last_updated":"2019-02-07T00:27:53Z","snapshot_observed_at":"2026-07-06T07:30:26.236692Z","submitted_at":"2019-01-31T23:59:34Z","title":"Tsallis Reinforcement Learning: A Unified Framework for Maximum Entropy Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"1902.00137","doi":null,"metadata_source":"pith","pith_arxiv_id":"1902.00137","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tsallis Reinforcement Learning: A Unified Framework for Maximum Entropy Reinforcement Learning","venue":"cs.LG","work_id":"afb5302b-a5d7-4be7-b543-ca6cd662e649","year":2019},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"cited_paper":"/paper/1902.00137","citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:0c41623d8ba34603cb8b307d8e7837df787cdf0043963a8c3a4eeb6eab2bf6f9","observation_id":"193ea8ce-7727-4804-8cdf-9240efd3a30e","resolution":{"observed_at":"2026-05-25T01:35:10.843852Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sparse Markov decision processes with causal sparse Tsallis entropy regularization for reinforcement learning","venue":null,"work_id":"e596d72f-284c-4a37-9a97-3c64366bcf37","year":2018},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:644497c6269eec0ae5e4b076b49c932b5bee36655594537d9f1fdba645aafbe2","observation_id":"6387c191-33f1-48ee-8b54-82aef60790bb","resolution":{"observed_at":"2026-05-25T01:35:11.509716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Maximum Causal Tsallis Entropy Imitation Learning","venue":null,"work_id":"585c374b-6e7b-4bd4-b568-b15fb0f7bd10","year":2018},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:d7c9db153a324575029f5b164d5d77dc633c01734ddee83041451705355c67a9","observation_id":"4807cc18-ae8a-4f03-8b6c-931b2b527904","resolution":{"observed_at":"2026-05-25T01:35:11.415216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1405.6757","last_updated":"2014-05-26T23:11:40Z","snapshot_observed_at":"2026-07-06T03:44:44.165995Z","submitted_at":"2014-05-26T23:11:40Z","title":"Proximal Reinforcement Learning: A New Theory of Sequential Decision Making in Primal-Dual Spaces","version":1},"cited_work":{"arxiv_id":"1405.6757","doi":null,"metadata_source":"pith","pith_arxiv_id":"1405.6757","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proximal Reinforcement Learning: A New Theory of Sequential Decision Making in Primal-Dual Spaces","venue":"cs.LG","work_id":"a8edc7e4-3d77-4e0a-ac8f-b7221c70e0f7","year":2014},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"cited_paper":"/paper/1405.6757","citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:cbf0a810661514fb2e538667311b4f5c4f3a0a00228a740c00d2e25757720ccd","observation_id":"2147e710-cf1a-4061-ab3b-21a137bcffa9","resolution":{"observed_at":"2026-05-25T01:35:10.800991Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Markov processes and the H-theorem","venue":null,"work_id":"5ba82388-aac1-4072-b807-db8bc2bd7fd2","year":1963},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:07e0a0eae077a8f32cc3ad01de78486245e2036e1f9c569ee0c8c144119c5feb","observation_id":"63b70ca9-abbe-4542-858a-dd71d889e830","resolution":{"observed_at":"2026-05-25T01:35:11.574574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A General Class of Coefﬁcients of Divergence of One Distribution from Another","venue":null,"work_id":"07ced6ef-f746-4350-a8b3-d09b61cbf550","year":1966},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:7db6e9370788cdec826ec10cf507ffc64cba9b87da3c2507a9b5e176fb3fec56","observation_id":"4bab8e75-5611-40f5-a7b6-2cceec3e9131","resolution":{"observed_at":"2026-05-25T01:35:11.411930Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Convex Optimization; Cambridge University Press: Cambridge, UK, 2004; 487p","venue":null,"work_id":"735a1c7b-07cf-4bb0-9528-bd3afab0c961","year":2004},"citing_paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-25T01:34:29.047426Z"},"links":{"citing_paper":"/paper/1907.04214"},"observation_digest":"sha256:5b29d7103968269e705c0e363ebef20252624be0a56a3248df9392b3e18058fb","observation_id":"d646aca9-c28e-40dd-8ff2-cb314217a33e","resolution":{"observed_at":"2026-05-25T01:35:11.505146Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"1907.04214","last_updated":"2019-07-18T07:12:31Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-06T15:02:56Z","title":"Entropic Regularization of Markov Decision Processes"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":11,"verified_fuzzy":40},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 0 inbound Pith citation observations for arXiv:1907.04214."}