{"as_of":"2026-08-10T01:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e206771d946bafed476495d70d4d58c23a664f0f0e6c25b69d9604442b885a6e","coverage":[{"denominator":13,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":13,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-26T21:35:56.749428Z","state":"measured"},{"denominator":13,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":13,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.18963/citation-record","integrity":"/paper/2606.18963/integrity","json":"/paper/2606.18963/citation-record.json","paper":"/paper/2606.18963"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1606.01540","last_updated":"2016-06-05T17:54:48Z","snapshot_observed_at":"2026-08-09T23:24:21.399948Z","submitted_at":"2016-06-05T17:54:48Z","title":"OpenAI Gym","version":1},"cited_work":{"arxiv_id":"1606.01540","doi":"10.1109/jssc.2019","metadata_source":"pith","pith_arxiv_id":"1606.01540","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"OpenAI Gym","venue":"cs.LG","work_id":"6af98f3f-f074-41ae-a689-7dd7b4b8efde","year":2016},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"cited_paper":"/paper/1606.01540","citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:65e57627213538c4fe87b6972a50a79a638d7ed10903da0e4e5feaeedb818d36","observation_id":"ca3b200f-43e2-4324-8d44-ad9a2de8f189","resolution":{"observed_at":"2026-07-03T23:59:06.821564Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T21:35:56.749428Z","title":null,"venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:165f062c2d0243651c31e20087f5487814408e1d2715e364a556a60ac1a57a03","observation_id":"c131fe19-cb3f-42f1-ae28-2d273b00862c","resolution":{"observed_at":"2026-06-26T21:35:56.749428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04104","last_updated":"2024-04-17T17:41:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-01-10T18:12:16Z","title":"Mastering Diverse Domains through World Models","version":2},"cited_work":{"arxiv_id":"2301.04104","doi":"10.1126/sciadv.adu2488","metadata_source":"pith","pith_arxiv_id":"2301.04104","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Diverse Domains through World Models","venue":"cs.AI","work_id":"6aeb260f-8c7c-4f9c-b98b-067cd7c59acd","year":2023},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"cited_paper":"/paper/2301.04104","citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:4317e557f93d5a51ba48a31307a81ca1db1e5a62b262464cc993ec7613bd6138","observation_id":"1edb5864-fa7f-45c6-9663-99f9a9f891c4","resolution":{"observed_at":"2026-07-03T23:59:06.813938Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T21:35:56.749428Z","title":"InNeurIPS","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:cef27ffcca730d3d57f5b3ead633f5e4f00e2ef92d4ef54cb61ad2c65d4428ff","observation_id":"65b74b29-1dce-4bb5-9888-b5668aef2909","resolution":{"observed_at":"2026-06-26T21:35:56.749428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T21:35:56.749428Z","title":"Laskin,M.;Yarats,D.;Liu,H.;Lee,K.;Zhan,A.;Lu,K.;Cang,C.; Pinto,L.;andAbbeel,P.2021.URLB:UnsupervisedReinforcement Learning Benchmark","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:06dbdb3fc4f9354bd55babb1ae2944b037b128ad00290d5815afb947baaed5e0","observation_id":"d2828c0a-d737-4bf8-86c1-abbde782a35b","resolution":{"observed_at":"2026-06-26T21:35:56.749428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:276844ab598866d0bfd00096c496f46f4b73d6339cc89ef5168f28e20002e6e0","observation_id":"da6872f2-06bc-433a-a3ba-830a52e4cc69","resolution":{"observed_at":"2026-07-03T23:59:06.825532Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T21:35:56.749428Z","title":null,"venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:4d40eb4adea60b9ecf4cf2b1b1101e33f96f92a6d36659b13109fd94f9b6c42b","observation_id":"82580466-26f0-49e7-8125-16e8983d5fc9","resolution":{"observed_at":"2026-06-26T21:35:56.749428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.18701","last_updated":"2026-06-16T03:01:32Z","snapshot_observed_at":"2026-07-06T23:05:30.879164Z","submitted_at":"2026-04-20T18:01:15Z","title":"Curiosity-Critic: Cumulative Prediction Error Improvement as a Tractable Intrinsic Reward for World Model Training","version":3},"cited_work":{"arxiv_id":"2604.18701","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.18701","snapshot_observed_at":"2026-07-03T23:59:06.815298Z","title":"Curiosity-Critic: Cumulative Prediction Error Improvement as a Tractable Intrinsic Reward for World Model Training","venue":"cs.LG","work_id":"bf5adeb8-dcdc-4523-b23e-fc3a1207b1f0","year":2026},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"cited_paper":"/paper/2604.18701","citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:d6abec7b6517c32b14248b976f6ae892f140f726dabca1ddbb1fe15bcd1547bf","observation_id":"c2b922f3-844e-479f-b741-7cb72cba76e8","resolution":{"observed_at":"2026-07-03T23:59:06.817201Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10342","last_updated":"2024-07-15T04:19:50Z","snapshot_observed_at":"2026-08-09T14:11:33.314722Z","submitted_at":"2024-02-15T22:11:18Z","title":"Exploration-Driven Policy Optimization in RLHF: Theoretical Insights on Efficient Data Utilization","version":2},"cited_work":{"arxiv_id":"2402.10342","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.10342","snapshot_observed_at":"2026-07-03T23:59:06.798546Z","title":"Exploration-driven policy optimization in rlhf: Theoretical insights on efficient data utilization","venue":null,"work_id":"c60ca356-dae0-40c1-a581-81836e094187","year":2025},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"cited_paper":"/paper/2402.10342","citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:10cc84fcd1426017488c36f2ce35dc3b3e9b7e52be9f67bfc43ee6ab573986df","observation_id":"c95c199c-91b2-4ff6-a7dc-76749bedb1da","resolution":{"observed_at":"2026-07-03T23:59:06.801171Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T21:35:56.749428Z","title":"Puterman, M","venue":null,"work_id":null,"year":1994},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:2d4067b54d6ec6681e351a3623c665d0df36e737c92013c178cba3e3d54ea242","observation_id":"be168cc3-1723-4b87-87da-4155972264a2","resolution":{"observed_at":"2026-06-26T21:35:56.749428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T21:35:56.749428Z","title":"Sutton, R","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:00ce20a34fc4d349f8ddad66fb307f0d53ab935ca82ed086916acf0eed8b0f0d","observation_id":"ac2da2a3-9878-4488-94dc-71a4b36bb884","resolution":{"observed_at":"2026-06-26T21:35:56.749428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.07473","last_updated":"2024-05-13T05:18:23Z","snapshot_observed_at":"2026-07-30T11:49:51.806264Z","submitted_at":"2024-05-13T05:18:23Z","title":"Intrinsic Rewards for Exploration without Harm from Observational Noise: A Simulation Study Based on the Free Energy Principle","version":1},"cited_work":{"arxiv_id":"2405.07473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.07473","snapshot_observed_at":"2026-07-03T23:59:06.806588Z","title":"Wagenmaker, A.; Chen, Y.; Simchowitz, M.; Du, S","venue":null,"work_id":"86025cb0-d015-4fbc-b1f1-d46565c14ce9","year":null},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"cited_paper":"/paper/2405.07473","citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:d7bf7d44b65c06d0a76d3831662f1eb7324b0e96bcdbe59e2a20b51cda582d09","observation_id":"a84a3ed1-7807-429a-a2c1-7d46b22aef8c","resolution":{"observed_at":"2026-07-03T23:59:06.808701Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19548","last_updated":"2025-04-25T01:53:37Z","snapshot_observed_at":"2026-07-06T18:22:18.644161Z","submitted_at":"2024-05-29T22:23:20Z","title":"RLeXplore: Accelerating Research in Intrinsically-Motivated Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2405.19548","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.19548","snapshot_observed_at":"2026-07-03T23:59:06.802881Z","title":null,"venue":null,"work_id":"3a0b38fc-75ff-452e-b075-d1864866fc30","year":2024},"citing_paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:56.749428Z"},"links":{"cited_paper":"/paper/2405.19548","citing_paper":"/paper/2606.18963"},"observation_digest":"sha256:3e62fc547b05b7606a6a4e6f86c824cf1c4ba5cd9e33b88e2429f69e21e156a6","observation_id":"fadd1b62-43d5-4422-b18e-e11d56cafc0f","resolution":{"observed_at":"2026-07-03T23:59:06.805552Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2606.18963","last_updated":"2026-06-17T11:43:10Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-31T13:51:11.220261Z","submitted_at":"2026-06-17T11:43:10Z","title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards"},"reference_resolution":{"displayed":13,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":6,"verified_exact":5,"verified_fuzzy":0},"total_outbound_references":13},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 13 of 13 outbound references and 0 inbound Pith citation observations for arXiv:2606.18963."}