{"as_of":"2026-08-08T22:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:da72b67d291e619f91297590318ca8f5de820bd0a64c109f3693c5921807a507","coverage":[{"denominator":20,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:58:21.440862Z","state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.04187/citation-record","integrity":"/paper/2507.04187/integrity","json":"/paper/2507.04187/citation-record.json","paper":"/paper/2507.04187"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:58:23.497013Z","title":"C.2 Treatment Allocation for Sepsis Patients We utilize the MIMIC-III Clinical Database to construct our environment for Sepsis patients","venue":null,"work_id":"2bbb25f0-8ddb-4bed-a82c-e77faf34b6dc","year":2025},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:21.134632Z"},"links":{"citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:52a51de2339ef9a0e47f70b2f82eac5040c75dca45c6269275cddbf2a951b78e","observation_id":"9df342b2-8219-4261-af34-b8175019f888","resolution":{"observed_at":"2026-08-06T19:58:23.609894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:58:23.012067Z","title":"This condition is typically met by standard tabular machine learning algorithms","venue":null,"work_id":"d11f011d-7851-4340-8d60-c2fffa406af6","year":2018},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:21.335508Z"},"links":{"citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:15c6d38cdade391986ba7e0fc5cc276db0ec46af8233fd7179dc90427bf2858d","observation_id":"e8bf660f-5299-4e58-8a57-fafc58285a48","resolution":{"observed_at":"2026-08-06T19:58:23.131846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1903.00374","last_updated":"2024-04-03T14:26:32Z","snapshot_observed_at":"2026-08-07T15:39:37.899352Z","submitted_at":"2019-03-01T15:40:19Z","title":"Model-Based Reinforcement Learning for Atari","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1903.00374","snapshot_observed_at":"2026-08-06T19:58:19.733463Z","title":"Model-based reinforcement learning for atari","venue":null,"work_id":null,"year":1903},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:19.733463Z"},"links":{"cited_paper":"/paper/1903.00374","citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:3900d9efd630a89dda4bd6daf7cabb6db717388be18d3a2bfd804a83f136f479","observation_id":"ec79a4de-c4d4-480d-a0eb-a4465a56a4b2","resolution":{"observed_at":"2026-08-06T19:58:19.733463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.08940","last_updated":"2023-10-02T00:55:29Z","snapshot_observed_at":"2026-08-01T18:59:39.993355Z","submitted_at":"2023-01-21T11:30:13Z","title":"Quasi-optimal Reinforcement Learning with Continuous Actions","version":2},"cited_work":{"arxiv_id":"2301.08940","doi":null,"metadata_source":"pith","pith_arxiv_id":"2301.08940","snapshot_observed_at":"2026-08-06T19:58:22.263099Z","title":"Quasi-optimal Reinforcement Learning with Continuous Actions","venue":"stat.ML","work_id":"ffd26e1b-6b5c-47fa-a6b9-855e2fef6878","year":2023},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:20.217655Z"},"links":{"cited_paper":"/paper/2301.08940","citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:412a9d931dd5a2066de677716edfbede07762026c75b9a445271ba0688a71396","observation_id":"f9266fcf-3437-40ca-a0a5-1007293cd133","resolution":{"observed_at":"2026-08-06T19:58:22.385863Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.14281","last_updated":"2024-07-30T15:42:59Z","snapshot_observed_at":"2026-08-01T10:54:57.346974Z","submitted_at":"2023-03-24T21:39:06Z","title":"Sequential Knockoffs for Variable Selection in Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2303.14281","doi":null,"metadata_source":"pith","pith_arxiv_id":"2303.14281","snapshot_observed_at":"2026-08-06T19:58:22.021917Z","title":"Sequential Knockoffs for Variable Selection in Reinforcement Learning","venue":"stat.ML","work_id":"21556aa3-22a8-4a6b-a893-08a0050d2af2","year":2023},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:20.447001Z"},"links":{"cited_paper":"/paper/2303.14281","citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:987e289d203590e4883b515fad324a55e3a755d6e7b13b94f76b175fa5ad98a0","observation_id":"f7c6235a-514c-4b58-9092-c03351d87ec6","resolution":{"observed_at":"2026-08-06T19:58:22.102178Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-06T19:58:20.695847Z","title":"Proximal policy optimization algorithms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:20.695847Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:e1ec76cebd02398452b0e0c1b21d191af7c4639c9550d00ca974541a93330fd4","observation_id":"2b4f1771-5967-4d9a-8cd9-0fd7c74dfb36","resolution":{"observed_at":"2026-08-06T19:58:20.695847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.06781","last_updated":"2020-01-19T06:07:20Z","snapshot_observed_at":"2026-07-06T08:51:25.168612Z","submitted_at":"2020-01-19T06:07:20Z","title":"FRESH: Interactive Reward Shaping in High-Dimensional State Spaces using Human Feedback","version":1},"cited_work":{"arxiv_id":"2001.06781","doi":null,"metadata_source":"pith","pith_arxiv_id":"2001.06781","snapshot_observed_at":"2026-08-06T19:58:21.615308Z","title":"FRESH: Interactive Reward Shaping in High-Dimensional State Spaces using Human Feedback","venue":"cs.AI","work_id":"2658affb-192d-48a3-85c6-4ab98ea42f48","year":2020},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:20.970823Z"},"links":{"cited_paper":"/paper/2001.06781","citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:d2157e1d2d537c6a94589b0fef44dc6d88de7e9896b593556cd489c51f7594f5","observation_id":"71d44aa7-5877-4d8c-92cb-277a1e0e7ac9","resolution":{"observed_at":"2026-08-06T19:58:21.676126Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:58:23.287084Z","title":null,"venue":null,"work_id":"3c899599-b250-4334-8e79-d0b4415ea70a","year":2025},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:21.231112Z"},"links":{"citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:3420601361aba8cba6d0aac80b92f3e2f0b144ce536f4c6c8b728fb90825985f","observation_id":"afd1b8d0-f784-4674-ad9a-4e02b89a5410","resolution":{"observed_at":"2026-08-06T19:58:23.384146Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:58:22.755571Z","title":"Then for suchϵ, denote Ω :={i :ϵi =−1}, which is a subset ofH0 by the assumption (and recall thatH0 is the collection of all null variables)","venue":null,"work_id":"3e12757e-5c38-45b3-8a73-dd8b4fd43a87","year":2015},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:21.440862Z"},"links":{"citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:411fdc3fce823a9ec81004138720a7300f7e9a3d38eb9a38474b5495b1328a30","observation_id":"790f878e-7e0c-4988-9cee-407c0419c501","resolution":{"observed_at":"2026-08-06T19:58:22.869856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1202.3725","last_updated":"2012-02-14T16:41:17Z","snapshot_observed_at":"2026-08-04T03:11:14.596001Z","submitted_at":"2012-02-14T16:41:17Z","title":"Generalized Fisher Score for Feature Selection","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1202.3725","snapshot_observed_at":"2026-08-06T19:58:19.447628Z","title":"Generalized fisher score for feature selection.arXiv preprint arXiv:1202.3725,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2009,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:19.447628Z"},"links":{"cited_paper":"/paper/1202.3725","citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:0da52a73e5de29b5a3ecb23f48ebadb131573ada8266bdc826b59d2c8b3d9676","observation_id":"9378e9b2-1a72-447f-8335-7fa9a1e3eff7","resolution":{"observed_at":"2026-08-06T19:58:19.447628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1512.01124","last_updated":"2015-12-16T17:34:55Z","snapshot_observed_at":"2026-07-06T04:38:38.910342Z","submitted_at":"2015-12-03T15:51:30Z","title":"Deep Reinforcement Learning with Attention for Slate Markov Decision Processes with High-Dimensional States and Actions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1512.01124","snapshot_observed_at":"2026-08-06T19:58:20.864765Z","title":"Deep reinforcement learning with attention for slate markov decision processes with high-dimensional states and actions","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2011,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:20.864765Z"},"links":{"cited_paper":"/paper/1512.01124","citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:aa5c87bbc589b8b780b917523dc3c2061aca53e6c29a2b74e9132d5c530d62fe","observation_id":"31e1abca-1c61-4ab8-83b0-a0716e62a01e","resolution":{"observed_at":"2026-08-06T19:58:20.864765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1703.03454","last_updated":"2017-03-09T20:22:27Z","snapshot_observed_at":"2026-07-06T05:33:06.530094Z","submitted_at":"2017-03-09T20:22:27Z","title":"Sample Efficient Feature Selection for Factored MDPs","version":1},"cited_work":{"arxiv_id":"1703.03454","doi":null,"metadata_source":"pith","pith_arxiv_id":"1703.03454","snapshot_observed_at":"2026-08-06T19:58:22.494684Z","title":"Sample Efficient Feature Selection for Factored MDPs","venue":"cs.LG","work_id":"572cbe63-67c2-4498-a49b-2bb8e2f02a86","year":2017},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2012,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:19.576991Z"},"links":{"cited_paper":"/paper/1703.03454","citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:0c75aebdff56c896b0d27f8bbde6dd7ddb205ca63ca96e8521a9223fbc5ca745","observation_id":"b809d0b4-1ece-45db-b4af-c2a9d1f658af","resolution":{"observed_at":"2026-08-06T19:58:22.570256Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:58:24.167099Z","title":"Modern perspectives on reinforcement learning in finance.Modern Perspectiveson ReinforcementLearning in Finance (September 6, 2019)","venue":null,"work_id":"a18bf8db-d16a-4fb9-9d20-26cfa843746e","year":2019},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2013,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:19.989486Z"},"links":{"citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:df4aa37f557c5adbfa8b480db085a6a56d3f495896bd0300d2056ff042b74765","observation_id":"b228eede-2d25-4c27-b197-7cad3bac34da","resolution":{"observed_at":"2026-08-06T19:58:24.254348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:58:24.509606Z","title":"Growing action spaces","venue":null,"work_id":"518b5d3d-2059-48ce-8b2b-e37b73caa824","year":2025},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:19.344036Z"},"links":{"citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:811ec0d8d3d31a0fe9f2e2d6870aeae0d1ceeb6f3d486b827dc9889193be1a41","observation_id":"16e60d3e-f86f-49f5-b7a5-fbdd0d0b02b8","resolution":{"observed_at":"2026-08-06T19:58:24.587546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:58:24.354748Z","title":"Action space shaping in deep reinforcement learning","venue":null,"work_id":"9342919c-71a1-4835-b02a-ca1a3346106b","year":2020},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:19.856059Z"},"links":{"citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:90f95573c556ec0c04df05f3e5fb278df99e858e49a8dd7db04a54aacbd0ab5c","observation_id":"56d90c08-b572-469f-96eb-ba29dc9da6cd","resolution":{"observed_at":"2026-08-06T19:58:24.408502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1312.5602","last_updated":"2013-12-19T16:00:08Z","snapshot_observed_at":"2026-07-06T03:31:23.521122Z","submitted_at":"2013-12-19T16:00:08Z","title":"Playing Atari with Deep Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1312.5602","snapshot_observed_at":"2026-08-06T19:58:20.548361Z","title":"Playing atari with deep reinforcement learning.arXiv preprint arXiv:1312.5602,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:20.548361Z"},"links":{"cited_paper":"/paper/1312.5602","citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:64bd0d41175a8604aa1ea8edb9e649c529b193cb01982aa3d311a4131653903e","observation_id":"94004926-3825-4132-9d68-ab21500bb345","resolution":{"observed_at":"2026-08-06T19:58:20.548361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:58:24.031875Z","title":"Deep reinforcement learning in continuous action spaces: a case study in the game of simulated curling","venue":null,"work_id":"69788e24-dbfc-45d1-be78-d919e2976ecf","year":2025},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:20.102586Z"},"links":{"citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:ac3b25ba4cbc7fe806896b565c5d5c38ff8449b310100f704e429cb3b285cbf8","observation_id":"602f891a-aa8c-49bb-874c-e95ad1ecf200","resolution":{"observed_at":"2026-08-06T19:58:24.078551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.10765","last_updated":"2018-09-27T21:05:19Z","snapshot_observed_at":"2026-08-04T04:20:14.294720Z","submitted_at":"2018-09-27T21:05:19Z","title":"Auto-Encoding Knockoff Generator for FDR Controlled Variable Selection","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.10765","snapshot_observed_at":"2026-08-06T19:58:20.357047Z","title":"Auto-encoding knockoff generator for fdr controlled variable selection.arXiv preprint arXiv:1809.10765,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:20.357047Z"},"links":{"cited_paper":"/paper/1809.10765","citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:0c53ab1a4763550734523fe93ff966dac7a72b45cfd45e550aa8344fc0900f2a","observation_id":"072990f4-7d05-4765-aeb1-0fd3c5c074a4","resolution":{"observed_at":"2026-08-06T19:58:20.357047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1706.04677","last_updated":"2017-06-14T21:42:12Z","snapshot_observed_at":"2026-07-31T13:51:02.653968Z","submitted_at":"2017-06-14T21:42:12Z","title":"Gene Hunting with Knockoffs for Hidden Markov Models","version":1},"cited_work":{"arxiv_id":"1706.04677","doi":null,"metadata_source":"pith","pith_arxiv_id":"1706.04677","snapshot_observed_at":"2026-08-06T19:58:21.783198Z","title":"Gene Hunting with Knockoffs for Hidden Markov Models","venue":"stat.ME","work_id":"3a712720-2671-49d2-bdaf-9a5f15962089","year":2017},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:20.777883Z"},"links":{"cited_paper":"/paper/1706.04677","citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:e340e31fb5faa28718084ac8886b86714a30253d051c1fa1e6d6203efe031fb9","observation_id":"caf74bf3-4425-4c58-8be6-17dff894d6c5","resolution":{"observed_at":"2026-08-06T19:58:21.857131Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:58:23.752263Z","title":"(2023) adopted a two-stage framework, performing variable selection offline before applying reinforcement learning","venue":null,"work_id":"a09d232e-fbe8-4a6d-8d1d-a9393c2c344b","year":2023},"citing_paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T19:58:21.040154Z"},"links":{"citing_paper":"/paper/2507.04187"},"observation_digest":"sha256:3c71b77a1ae98e369eecd97a7a4d4150254ae7b9092ce5ae71e9cecc019aec26","observation_id":"f5db8414-e75c-4636-956e-1f35771c0f4f","resolution":{"observed_at":"2026-08-06T19:58:23.886205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.04187","last_updated":"2025-07-05T23:40:55Z","latest_version":1,"primary_category":"stat.ML","snapshot_observed_at":"2026-08-06T19:50:53.867155Z","submitted_at":"2025-07-05T23:40:55Z","title":"Where to Intervene: Action Selection in Deep Reinforcement Learning"},"reference_resolution":{"displayed":20,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":7,"verified_exact":4,"verified_fuzzy":8},"total_outbound_references":20},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 20 of 20 outbound references and 0 inbound Pith citation observations for arXiv:2507.04187."}