{"as_of":"2026-08-09T00:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:476d8b060b62fd4661072e7bcd3a460b7380017b833fe369881b2dbe13d84b44","coverage":[{"denominator":33,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:59:16.541167Z","state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.01470/citation-record","integrity":"/paper/2507.01470/integrity","json":"/paper/2507.01470/citation-record.json","paper":"/paper/2507.01470"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1901.08492","last_updated":"2019-01-24T16:44:16Z","snapshot_observed_at":"2026-07-06T07:28:48.758265Z","submitted_at":"2019-01-24T16:44:16Z","title":"Feudal Multi-Agent Hierarchies for Cooperative Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1901.08492","snapshot_observed_at":"2026-08-06T20:59:13.632344Z","title":"Feudal multi-agent hierarchies for cooperative reinforcement learning, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.632344Z"},"links":{"cited_paper":"/paper/1901.08492","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:ac265427aa2fe1b1346079671471c964e169591b44bb64a751a8e1a9cee59a6d","observation_id":"7d4da9db-1c46-43ae-a885-322a86a7f863","resolution":{"observed_at":"2026-08-06T20:59:13.632344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1606.06565","last_updated":"2016-07-25T17:23:29Z","snapshot_observed_at":"2026-07-06T05:00:46.434335Z","submitted_at":"2016-06-21T13:37:05Z","title":"Concrete Problems in AI Safety","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.06565","snapshot_observed_at":"2026-08-06T20:59:13.697707Z","title":"Concrete Problems in AI Safety , July 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.697707Z"},"links":{"cited_paper":"/paper/1606.06565","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:f3ebf5667531cd380330bbab02d8aef440a52abd5bfd69b9662c6320f831f01e","observation_id":"99c0bd26-131a-419b-b279-4e8d9470bc95","resolution":{"observed_at":"2026-08-06T20:59:13.697707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:19.514654Z","title":"Hindsight experience replay","venue":null,"work_id":"f0bdaef6-ca0c-4c45-80e3-19a03b151aff","year":2017},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.773390Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:9786d1a3871f647a6e44e677cdea58911bf69dda15bd6090c0375e46746340c5","observation_id":"8836cee5-a8d6-4999-a299-a65dc61cdfee","resolution":{"observed_at":"2026-08-06T20:59:19.581118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:19.395720Z","title":"Dynamic programming","venue":null,"work_id":"17dc7a8f-efad-4bf5-93d5-f6ddea51f80f","year":1957},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.828930Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:9bedb7176270e8e84d22ec0d41769cd9ac95974e688451c0bc1ce58de7cdfa29","observation_id":"14ad7b67-bf52-4977-83fa-a00466b041a1","resolution":{"observed_at":"2026-08-06T20:59:19.452897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.12894","last_updated":"2018-10-30T17:44:42Z","snapshot_observed_at":"2026-07-06T07:11:32.319931Z","submitted_at":"2018-10-30T17:44:42Z","title":"Exploration by Random Network Distillation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.12894","snapshot_observed_at":"2026-08-06T20:59:13.902535Z","title":"Exploration by Random Network Distillation , October 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.902535Z"},"links":{"cited_paper":"/paper/1810.12894","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:cf615138cba4f24b1fb0bb836d56483f8d009dc6987664e3db60cd513cd9f2d5","observation_id":"e4fb16a7-b28d-410a-9931-4d125ab8aa59","resolution":{"observed_at":"2026-08-06T20:59:13.902535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:13.972079Z","title":null,"venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:13.972079Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:ec240d07493897eb0e162aaf173f43ff404507f052bf4125d5b276e183fb98b1","observation_id":"bc377857-9684-4efc-b592-66816d3ead45","resolution":{"observed_at":"2026-08-06T20:59:13.972079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2024.12411","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.496695Z","title":"HiSOMA : A hierarchical multi-agent model integrating self-organizing neural networks with multi-agent deep reinforcement learning","venue":null,"work_id":"6a22b63a-6f71-40d0-a0c4-55ee5cb672bc","year":2024},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.067851Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:3179f65ac13e4481b90a03863eb8b48b83f9572cad22d1f254ceaa1d72c19d1c","observation_id":"5bfdeb87-e46b-4126-ac23-e2f7d48435d9","resolution":{"observed_at":"2026-08-06T20:59:17.554700Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:19.235043Z","title":"MASER : Multi-agent reinforcement learning with subgoals generated from experience replay buffer","venue":null,"work_id":"c6fffe30-35d0-4ec3-8293-646e6a25a360","year":2022},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.151775Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:976ecd2c4fc71b8a037803534c045c1b708aa7bb9db6a2e9f1635387fda9a40b","observation_id":"dde65c51-d02b-4875-bbc3-adef23066d2a","resolution":{"observed_at":"2026-08-06T20:59:19.333627Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:19.154172Z","title":"Automatic discovery of subgoals in reinforcement learning using strongly connected components","venue":null,"work_id":"c51f28ab-91d6-4763-ba86-39e81eb529f9","year":2009},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.229955Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:5696ab0014ac0001b165cf7b628b3394cd45ef7dcb0cc932c7f6fe487252715e","observation_id":"d311e481-1434-40ba-9ab9-c61e70c41aed","resolution":{"observed_at":"2026-08-06T20:59:19.194182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:14.308450Z","title":"Exploration in deep reinforcement learning: A survey","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.308450Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:0cba49a3a2d934d3a8c53c9689af8aff43e66abb787d7fc318fe378c0281b488","observation_id":"467cdcee-7649-43a1-8f56-49382aafe968","resolution":{"observed_at":"2026-08-06T20:59:14.308450Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:14.397566Z","title":"Lecun, L","venue":null,"work_id":null,"year":1998},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.397566Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:1723eaf6b3024e873be50a10d8bd201532e7e8186daab102e9a2c977b33a05c4","observation_id":"5c6a3b94-79d5-418b-9334-34cc91bef9c4","resolution":{"observed_at":"2026-08-06T20:59:14.397566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:19.039857Z","title":"Automatic discovery of subgoals in reinforcement learning using diverse density","venue":null,"work_id":"e4b218a1-8947-4e8d-93a0-6d530d4a2ccc","year":2001},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.518805Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:80b564c140ed7d85c81f265b69188198b733cb1385cbda050d8ab25f16b67cee","observation_id":"d9ce9363-1912-4b43-b219-5e02f2885fd2","resolution":{"observed_at":"2026-08-06T20:59:19.082703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.54097/er0mx710","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.163674Z","title":"Research on Multi -agent Sparse Reward Problem","venue":null,"work_id":"ef5e1eb5-d041-4b33-8cba-8caa6133e739","year":2024},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.604076Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:1ceae58fd97bdc83e0dc2194254720dd0640374bddca57306299a5f43051c0c6","observation_id":"5d82be80-cdee-46ff-818b-67cb0c575fd3","resolution":{"observed_at":"2026-08-06T20:59:17.239155Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:18.918288Z","title":"Laser learning environment: A new environment for coordination-critical multi-agent tasks","venue":null,"work_id":"12179465-535f-4fd4-a2ba-7f6e30804c13","year":2025},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.679103Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:1d544ada3aceb5fa968ccd9d75fa27c75104158ce1de6f4f0052845f9e6e1346","observation_id":"5a082f66-e373-4a5f-a84a-68e4806cc08a","resolution":{"observed_at":"2026-08-06T20:59:18.975027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1613/jair.1.14390","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:16.927232Z","title":"An overview of environmental features that impact deep reinforcement learning in sparse-reward domains","venue":null,"work_id":"c747ad70-7389-42c7-b36e-795f6b6a5a92","year":2023},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.776492Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:2a09245d8831712a5f5c09820d29b35bfa7ab5b085b537118e483c2b96143465","observation_id":"2c60f28c-eda0-4176-a4fa-434b66a51ee5","resolution":{"observed_at":"2026-08-06T20:59:17.060987Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:14.850478Z","title":"Efros, and Trevor Darrell","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.850478Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:9a8291aa587762e144e5989c889c617726579282f99e9766f6d7b4e8ab710a5f","observation_id":"ddab0dbb-b9fd-4f0d-800b-2f05df26db66","resolution":{"observed_at":"2026-08-06T20:59:14.850478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:18.672546Z","title":"Learning to Drive a Bicycle using Reinforcement Learning and Shaping","venue":null,"work_id":"85148589-e67f-4edc-b20e-67a57988eb17","year":1998},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:14.972744Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:c6ef038b8f0d659ace34b1491c566a171c74eaba75cda129a9413e0f284d655e","observation_id":"8c7f1459-afb7-4eca-8ca3-3382338a08f2","resolution":{"observed_at":"2026-08-06T20:59:18.795445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.11485","last_updated":"2018-06-06T17:58:09Z","snapshot_observed_at":"2026-08-06T12:03:41.509273Z","submitted_at":"2018-03-30T14:23:39Z","title":"QMIX: Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.11485","snapshot_observed_at":"2026-08-06T20:59:15.058602Z","title":"QMIX : Monotonic Value Function Factorisation for Deep Multi - Agent Reinforcement Learning","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.058602Z"},"links":{"cited_paper":"/paper/1803.11485","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:3089c2140469a09424c6a64d0dc3577f01219dd0b3afb698de8d4b7934a1167f","observation_id":"cbeecba4-df41-4b81-83a4-5ee1fa5b4c80","resolution":{"observed_at":"2026-08-06T20:59:15.058602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1902.04043","last_updated":"2019-12-09T07:26:52Z","snapshot_observed_at":"2026-07-06T07:32:24.974529Z","submitted_at":"2019-02-11T18:43:53Z","title":"The StarCraft Multi-Agent Challenge","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1902.04043","snapshot_observed_at":"2026-08-06T20:59:15.116702Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.116702Z"},"links":{"cited_paper":"/paper/1902.04043","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:4df861a9a9c3cab49a11a05d82c09c55b1649d7ec1b81653e153871d517a0857","observation_id":"3724e73d-6be2-4af4-8957-7a144c55dbcc","resolution":{"observed_at":"2026-08-06T20:59:15.116702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:18.438668Z","title":"Normalized cuts and image segmentation","venue":null,"work_id":"6e70bcff-f908-430c-8d3c-3aed37261f2d","year":2000},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.206690Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:ec43277fe2c26e39ed5488fb377d034507059a86eb21eaf4ddaabb7a24775108","observation_id":"5a62b7a4-3661-41f7-885b-ab7abe8a9ad7","resolution":{"observed_at":"2026-08-06T20:59:18.586251Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:15.274544Z","title":"Wolfe, and Andrew G","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.274544Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:d00afeee20ee32e411303a5c95e572aecb7408724b92a05777220d8168637c87","observation_id":"f0da5f00-7721-4c86-8dcd-a8d99e8b560c","resolution":{"observed_at":"2026-08-06T20:59:15.274544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:18.177805Z","title":"Leibo, Karl Tuyls, and Thore Graepel","venue":null,"work_id":"67ae067f-cf7f-40c5-9d87-0c4524004db9","year":2018},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.321206Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:3e85fdef17a2a8ad13e271fa25c5a21f723051fec00a2369d5d825cd3e1a3cde","observation_id":"082ccee4-2d28-4ccb-ab72-b3b636db67bd","resolution":{"observed_at":"2026-08-06T20:59:18.304632Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/3643852","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:16.777917Z","title":"Faster MIL -based subgoal identification for reinforcement learning by tuning fewer hyperparameters","venue":null,"work_id":"07f52d78-6d5d-4e2b-8e21-2697cfd817a8","year":2024},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.405171Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:b4cdb9d88d26ba2dd10022f3fa5d08048e7d001259f68db054fdd5ce44173841","observation_id":"6775740b-eb04-4378-a2ee-978b43fb2dca","resolution":{"observed_at":"2026-08-06T20:59:16.854696Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.995952Z","title":"Sutton and Andrew G","venue":null,"work_id":"aaf2cd12-0ed0-4071-984b-e07986803b97","year":2018},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.448124Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:379b9353d0158a68177ff153d8598129d74491160dcabafc056fa374c984382d","observation_id":"25b4aadd-313e-4b70-a586-c290970713a2","resolution":{"observed_at":"2026-08-06T20:59:18.088589Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:15.511885Z","title":"Sutton, Doina Precup, and Satinder Singh","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.511885Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:cc2d3089f98ba0db36987a0f906e5d36cd3eaea85971e5936b4a6455f69e694c","observation_id":"308c9bdf-4d87-4580-a124-1fbafbe0520f","resolution":{"observed_at":"2026-08-06T20:59:15.511885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.908255Z","title":"\\#exploration: A study of count-based exploration for deep reinforcement learning","venue":null,"work_id":"3928da3e-5842-4b98-b7ac-d8c4f6edd087","year":2017},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.649124Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:c7b84a74a1cb543eacb7152caf2da84d87a53f21dd288dcea6b29fbdddd27450","observation_id":"2c91e146-c36f-40b5-a796-6d9299277dbe","resolution":{"observed_at":"2026-08-06T20:59:17.954977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.795026Z","title":"Keeping your distance: Solving sparse reward tasks using self-balancing shaped rewards","venue":null,"work_id":"d961fbfb-ab76-46e1-9d92-1fb12e7c3610","year":2019},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.787024Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:553485d766e7dc34263c095dcafbc98de9e1ec86a092e79e2cbe9b8ef45347ad","observation_id":"8a892e58-5ae4-4b7c-bf72-9a0072e396e2","resolution":{"observed_at":"2026-08-06T20:59:17.854308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1509.06461","last_updated":"2015-12-08T21:19:16Z","snapshot_observed_at":"2026-08-07T14:09:19.448496Z","submitted_at":"2015-09-22T04:40:22Z","title":"Deep Reinforcement Learning with Double Q-learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1509.06461","snapshot_observed_at":"2026-08-06T20:59:15.925516Z","title":"Deep reinforcement learning with double Q - Learning","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:15.925516Z"},"links":{"cited_paper":"/paper/1509.06461","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:6bb5756a1f857745880cd370b139cf7493dd8da6b692b43857f9aab28e5a1c19","observation_id":"26d0f6c0-c6e6-46bb-b059-8b1e0516bb2b","resolution":{"observed_at":"2026-08-06T20:59:15.925516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2008.01062","last_updated":"2021-10-04T01:36:59Z","snapshot_observed_at":"2026-07-06T09:44:11.377445Z","submitted_at":"2020-08-03T17:52:09Z","title":"QPLEX: Duplex Dueling Multi-Agent Q-Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2008.01062","snapshot_observed_at":"2026-08-06T20:59:16.045425Z","title":"QPLEX : Duplex Dueling Multi - Agent Q - Learning , October 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:16.045425Z"},"links":{"cited_paper":"/paper/2008.01062","citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:4e94cbe472fc0cea89baafaee1d1960084ed1e756581fc82e630cd8e146b41b5","observation_id":"4e0992bd-c580-4fd7-870d-383f6305ef15","resolution":{"observed_at":"2026-08-06T20:59:16.045425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.691744Z","title":null,"venue":null,"work_id":"0977ecb7-f1e9-4155-b6bf-0d24354219b0","year":2009},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:16.145414Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:ef965d90444db2c447fb300342b3965a76887540ddb4617dba47ec433cec8fb8","observation_id":"2df939bb-a422-41e5-a98d-c16d404ec35e","resolution":{"observed_at":"2026-08-06T20:59:17.742184Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1609/aaai.v37i10.26386","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:16.636205Z","title":"HAVEN : Hierarchical cooperative multi-agent reinforcement learning with dual coordination mechanism","venue":null,"work_id":"2bff427b-47e1-46c7-a29f-2fe5cb43eb53","year":2023},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:16.303192Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:206b0ea12edeee6d12be55441fa9e91ffc1307f750a9fefeafbcbd918291b893","observation_id":"5f4be7ec-1b16-4267-b6c5-7c5eba4fd65b","resolution":{"observed_at":"2026-08-06T20:59:16.693495Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:17.643317Z","title":"Ng, Daishi Harada, and Stuart Russell","venue":null,"work_id":"d7f54022-8bdb-44dd-be86-bf07b1daff46","year":1999},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:16.392828Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:1eabf1b16ead5c1b841aa71d2718ae27521db39aeb009a70b41d736b22d29625","observation_id":"23accf97-5d55-4607-9472-1b9d15576a60","resolution":{"observed_at":"2026-08-06T20:59:17.668579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:59:16.541167Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T20:59:16.541167Z"},"links":{"citing_paper":"/paper/2507.01470"},"observation_digest":"sha256:84f83bdd8efebe2f97b665aed825cf71161a1497bfc7405daca25a9b9bb18250","observation_id":"ff4fd410-ad6b-4afd-aa4c-6643d5dac742","resolution":{"observed_at":"2026-08-06T20:59:16.541167Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.01470","last_updated":"2025-07-02T08:33:03Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-08T06:47:26.631798Z","submitted_at":"2025-07-02T08:33:03Z","title":"Zero-Incentive Dynamics: a look at reward sparsity through the lens of unrewarded subgoals"},"reference_resolution":{"displayed":33,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":15,"verified_exact":4,"verified_fuzzy":13},"total_outbound_references":33},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 33 of 33 outbound references and 0 inbound Pith citation observations for arXiv:2507.01470."}