{"as_of":"2026-08-08T08:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:50b7d33884085625f585424d1ac1d42e6a590fc165589479f97ab35a52582b75","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":11,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":11,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:02:48.448037Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T23:49:02.344960Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.05038","snapshot_observed_at":"2026-08-07T12:02:48.448037Z","title":"Recurrent model-free RL can be a strong baseline for many POMDPs,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.02050","last_updated":"2025-06-01T06:36:19Z","snapshot_observed_at":"2026-08-07T11:54:12.395783Z","submitted_at":"2025-06-01T06:36:19Z","title":"Decoupled Hierarchical Reinforcement Learning with State Abstraction for Discrete Grids","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:02:48.448037Z"},"links":{"cited_paper":"/paper/2110.05038","citing_paper":"/paper/2506.02050"},"observation_digest":"sha256:88fb36720511ef9b76a0b21bf2d8ff73bcea1724fa4451bd0d4205d71e1dbb0b","observation_id":"69a54379-6bff-43dd-99dd-511513960743","resolution":{"observed_at":"2026-08-07T12:02:48.448037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.05038","snapshot_observed_at":"2026-08-05T20:48:53.701392Z","title":"Recurrent model-free rl can be a strong baseline for many pomdps,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.09960","last_updated":"2025-08-13T17:28:39Z","snapshot_observed_at":"2026-08-06T07:55:29.959823Z","submitted_at":"2025-08-13T17:28:39Z","title":"GBC: Generalized Behavior-Cloning Framework for Whole-Body Humanoid Imitation","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-05T20:48:53.701392Z"},"links":{"cited_paper":"/paper/2110.05038","citing_paper":"/paper/2508.09960"},"observation_digest":"sha256:9daa5fd4e26becb32d5a62dc15c6d759c18e8fe079beb4764e354ed5618253e2","observation_id":"1bba64b6-349b-4b40-a804-ee262217b0ec","resolution":{"observed_at":"2026-08-05T20:48:53.701392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.05038","snapshot_observed_at":"2026-08-03T22:51:56.016926Z","title":"Recurrent model-free rl is a strong baseline for many POMDPs.arXiv preprint arXiv:2110.05038, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2511.08436","last_updated":"2026-07-19T21:08:10Z","snapshot_observed_at":"2026-08-07T15:28:21.927393Z","submitted_at":"2025-11-11T16:38:48Z","title":"Active Electrosensing and Communication in MARL-trained Weakly Electric Fish Collectives","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-03T22:51:56.016926Z"},"links":{"cited_paper":"/paper/2110.05038","citing_paper":"/paper/2511.08436"},"observation_digest":"sha256:aeaef8cd65d6c398f74c07ee0120a3105f1b721ec8696b8a6aaf34ed3964caa6","observation_id":"ac32d414-1f4a-4eb3-ab82-9f50314332d1","resolution":{"observed_at":"2026-08-03T22:51:56.016926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs","version":3},"cited_work":{"arxiv_id":"2110.05038","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2110.05038","snapshot_observed_at":"2026-07-03T23:49:02.344960Z","title":"Recurrent model-free rl can be a strong baseline for many pomdps","venue":null,"work_id":"2047c8c2-6c55-41a4-937c-16443e5e0c30","year":2021},"citing_paper":{"arxiv_id":"2602.05048","last_updated":"2026-05-01T20:59:20Z","snapshot_observed_at":"2026-08-03T04:28:51.084715Z","submitted_at":"2026-02-04T20:58:53Z","title":"MINT: Minimal Information Neuro-Symbolic Tree for Objective-Driven Knowledge-Gap Reasoning and Active Elicitation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T07:13:42.705619Z"},"links":{"cited_paper":"/paper/2110.05038","citing_paper":"/paper/2602.05048"},"observation_digest":"sha256:4c1c12356553e133e0adadeb071e86589b7867955c8f80c35b66e6ca07a3a477","observation_id":"5225a48a-9bbf-4634-ae6d-deddcbac7be3","resolution":{"observed_at":"2026-05-16T07:17:30.678350Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs","version":3},"cited_work":{"arxiv_id":"2110.05038","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2110.05038","snapshot_observed_at":"2026-07-03T23:49:02.344960Z","title":"Recurrent model-free rl can be a strong baseline for many pomdps","venue":null,"work_id":"2047c8c2-6c55-41a4-937c-16443e5e0c30","year":2021},"citing_paper":{"arxiv_id":"2602.19837","last_updated":"2026-05-06T07:57:45Z","snapshot_observed_at":"2026-07-06T22:46:45.213871Z","submitted_at":"2026-02-23T13:39:58Z","title":"Meta-Learning and Meta-Reinforcement Learning -- Tracing the Path towards DeepMind's Adaptive Agent","version":3},"reference_index":110,"source":"pdf_text","source_observed_at":"2026-05-15T20:46:15.275441Z"},"links":{"cited_paper":"/paper/2110.05038","citing_paper":"/paper/2602.19837"},"observation_digest":"sha256:ed23fe4a6a528d7322aea4bf3bec43e9c4e34dcfb7c1709bf94f8062ee670061","observation_id":"957fdfbf-93a6-49c2-9e88-592239812121","resolution":{"observed_at":"2026-05-15T20:46:35.740096Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.05038","snapshot_observed_at":"2026-08-02T21:30:17.383704Z","title":"arXiv:2110.05038 [cs]","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.20141","last_updated":"2026-05-28T15:55:29Z","snapshot_observed_at":"2026-08-04T06:27:14.421909Z","submitted_at":"2026-02-23T18:53:09Z","title":"Recurrent Structural Policy Gradient for Partially Observable Mean Field Games","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T21:30:17.383704Z"},"links":{"cited_paper":"/paper/2110.05038","citing_paper":"/paper/2602.20141"},"observation_digest":"sha256:6f1c25dcc4b896a0dfcabe5e3c8994d96aba77ae0ce7ec3e991efce0f069f9aa","observation_id":"24db73d1-e528-4a8c-b857-aa1dcdd88a04","resolution":{"observed_at":"2026-08-02T21:30:17.383704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs","version":3},"cited_work":{"arxiv_id":"2110.05038","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2110.05038","snapshot_observed_at":"2026-07-03T23:49:02.344960Z","title":"Recurrent model-free rl can be a strong baseline for many pomdps","venue":null,"work_id":"2047c8c2-6c55-41a4-937c-16443e5e0c30","year":2021},"citing_paper":{"arxiv_id":"2604.09671","last_updated":"2026-04-01T22:28:38Z","snapshot_observed_at":"2026-07-06T22:58:29.952947Z","submitted_at":"2026-04-01T22:28:38Z","title":"Belief-State RWKV for Reinforcement Learning under Partial Observability","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T21:56:09.642064Z"},"links":{"cited_paper":"/paper/2110.05038","citing_paper":"/paper/2604.09671"},"observation_digest":"sha256:23e029785d8a16e0f0ebccc8c6eabc9fb62cdf9029685c5f8791f87187547e19","observation_id":"ef71cc07-f3e7-44d3-8589-4e31c3b17b07","resolution":{"observed_at":"2026-05-13T21:58:19.937670Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs","version":3},"cited_work":{"arxiv_id":"2110.05038","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2110.05038","snapshot_observed_at":"2026-07-03T23:49:02.344960Z","title":"Recurrent model-free rl can be a strong baseline for many pomdps","venue":null,"work_id":"2047c8c2-6c55-41a4-937c-16443e5e0c30","year":2021},"citing_paper":{"arxiv_id":"2605.02552","last_updated":"2026-05-04T13:00:55Z","snapshot_observed_at":"2026-07-06T23:15:37.099197Z","submitted_at":"2026-05-04T13:00:55Z","title":"Recurrent Deep Reinforcement Learning for Chemotherapy Control under Partial Observability","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-08T18:51:16.753971Z"},"links":{"cited_paper":"/paper/2110.05038","citing_paper":"/paper/2605.02552"},"observation_digest":"sha256:bd433c0f18614e6a49bc4de0cdfd7993cd62ff3327a6004daa8daf541da10456","observation_id":"5ff5e595-5504-49ed-a64d-a8d53a613bdc","resolution":{"observed_at":"2026-05-09T06:10:38.523134Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs","version":3},"cited_work":{"arxiv_id":"2110.05038","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2110.05038","snapshot_observed_at":"2026-07-03T23:49:02.344960Z","title":"Recurrent model-free rl can be a strong baseline for many pomdps","venue":null,"work_id":"2047c8c2-6c55-41a4-937c-16443e5e0c30","year":2021},"citing_paper":{"arxiv_id":"2605.13740","last_updated":"2026-05-13T16:18:15Z","snapshot_observed_at":"2026-07-06T23:25:16.229144Z","submitted_at":"2026-05-13T16:18:15Z","title":"Learning POMDP World Models from Observations with Language-Model Priors","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-14T19:57:28.642081Z"},"links":{"cited_paper":"/paper/2110.05038","citing_paper":"/paper/2605.13740"},"observation_digest":"sha256:a26a4dffe2fd105c2ce1bf7e4fd9e5723556407672da16bf7e4d05c0d7fad7a4","observation_id":"9a3c8b48-9bf3-44c3-8636-152afeb3f5c2","resolution":{"observed_at":"2026-05-14T19:57:52.987389Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs","version":3},"cited_work":{"arxiv_id":"2110.05038","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2110.05038","snapshot_observed_at":"2026-07-03T23:49:02.344960Z","title":"Recurrent model-free rl can be a strong baseline for many pomdps","venue":null,"work_id":"2047c8c2-6c55-41a4-937c-16443e5e0c30","year":2021},"citing_paper":{"arxiv_id":"2606.18820","last_updated":"2026-06-17T08:41:55Z","snapshot_observed_at":"2026-08-07T14:10:41.321326Z","submitted_at":"2026-06-17T08:41:55Z","title":"Maturing Markov Decision Processes: Decision Making under Increasing Information and Shrinking Action Sets","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-26T21:47:16.490817Z"},"links":{"cited_paper":"/paper/2110.05038","citing_paper":"/paper/2606.18820"},"observation_digest":"sha256:004481dd14f60094d8b08c9bdfaf8cc96a7b0320a5d35347ddf00b41ecc0d9ff","observation_id":"1076a5da-7386-47bf-a13f-c3efd2ab27bb","resolution":{"observed_at":"2026-07-03T23:49:02.347183Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.05038","snapshot_observed_at":"2026-08-02T07:20:40.649329Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.11953","last_updated":"2026-07-15T13:05:53Z","snapshot_observed_at":"2026-08-08T05:16:44.265065Z","submitted_at":"2026-07-11T21:33:32Z","title":"When Does Reward Teach State? A Hidden-Automaton Instrument and the Group-Language Boundary","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-02T07:20:40.649329Z"},"links":{"cited_paper":"/paper/2110.05038","citing_paper":"/paper/2607.11953"},"observation_digest":"sha256:304997c81e6aaf46c1adc505b3cdee8a53ffd66b0d350f66a4cf43951e4bf2ff","observation_id":"aade19e8-2102-4b5c-b4e0-7afadf6e6cfe","resolution":{"observed_at":"2026-08-02T07:20:40.649329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2110.05038/citation-record","integrity":"/paper/2110.05038/integrity","json":"/paper/2110.05038/citation-record.json","paper":"/paper/2110.05038"},"outbound":[],"paper":{"arxiv_id":"2110.05038","last_updated":"2022-06-05T01:19:29Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T11:56:24.775449Z","submitted_at":"2021-10-11T07:09:14Z","title":"Recurrent Model-Free RL Can Be a Strong Baseline for Many POMDPs"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 11 inbound Pith citation observations for arXiv:2110.05038."}