{"as_of":"2026-08-18T05:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f7e66eeb2ce7a6e8f17cd243d52cef459e4beb01e0da87679d9396b162ebcfb8","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T14:57:30.265625Z","state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.03108/citation-record","integrity":"/paper/2608.03108/integrity","json":"/paper/2608.03108/citation-record.json","paper":"/paper/2608.03108"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2010.13611","last_updated":"2021-05-04T19:20:46Z","snapshot_observed_at":"2026-08-16T19:10:34.681472Z","submitted_at":"2020-10-26T14:31:08Z","title":"OPAL: Offline Primitive Discovery for Accelerating Offline Reinforcement Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.13611","snapshot_observed_at":"2026-08-15T14:57:30.167094Z","title":"ChenjiaBai, LingxiaoWang,Zhuoran Yang,ZhihongDeng, AnimeshGarg, PengLiu,and Zhaoran Wang","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.167094Z"},"links":{"cited_paper":"/paper/2010.13611","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:b4600ed0ea94ac4be1dd1b43914e8b685302c7b8373520b1de1d404002959a6d","observation_id":"31556764-0c0c-4591-a04c-39bbbfdc47b2","resolution":{"observed_at":"2026-08-15T14:57:30.167094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:57:30.186527Z","title":"Off-policy deep reinforcement learning without exploration","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.186527Z"},"links":{"citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:3d4dccdaabb55b53b28bdc403b2db96ce11fc02e2d313310eb7425d81ea8a8f4","observation_id":"bcdfc21f-8072-4a17-8f5e-09af7b5279b6","resolution":{"observed_at":"2026-08-15T14:57:30.186527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00321","last_updated":"2024-03-16T01:29:08Z","snapshot_observed_at":"2026-08-16T15:27:41.180475Z","submitted_at":"2023-06-01T03:36:06Z","title":"Improving Offline RL by Blending Heuristics","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00321","snapshot_observed_at":"2026-08-15T14:57:30.195104Z","title":"Improving offline rl by blending heuristics.arXiv preprint arXiv:2306.00321,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.195104Z"},"links":{"cited_paper":"/paper/2306.00321","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:cb5adc39a42dc1fcbf195a8fa31741197d717d3fa1de083878136ce0d5465360","observation_id":"44011897-b9d1-4779-869c-538b49e3bca3","resolution":{"observed_at":"2026-08-15T14:57:30.195104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.06169","last_updated":"2021-10-12T17:05:05Z","snapshot_observed_at":"2026-08-13T07:57:56.087944Z","submitted_at":"2021-10-12T17:05:05Z","title":"Offline Reinforcement Learning with Implicit Q-Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.06169","snapshot_observed_at":"2026-08-15T14:57:30.199368Z","title":"Offline reinforcement learning with implicit q- learning.arXiv preprint arXiv:2110.06169,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.199368Z"},"links":{"cited_paper":"/paper/2110.06169","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:cbf907d03ac415d8e3c3dd3b821d41a270bee2ce10da2ab3ef02728bf3cdaa2d","observation_id":"69e91176-85a2-4f7b-8e62-19d3034dd66a","resolution":{"observed_at":"2026-08-15T14:57:30.199368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:57:30.753084Z","title":"Conservative q-learning for offline reinforcementlearning.Advancesinneuralinformationprocessingsystems,33:1179–1191,2020","venue":null,"work_id":"28ceb627-293f-47cd-b570-cc7cd551bf45","year":2020},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.203753Z"},"links":{"citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:21773a57c301a325d95365526895ba32baeace9a543915fe3de6b40f2c70eda2","observation_id":"b3de7287-9c74-4cb8-bc26-486b000c7a2a","resolution":{"observed_at":"2026-08-15T14:57:30.757196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.01643","last_updated":"2020-11-01T23:50:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-04T17:00:15Z","title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.01643","snapshot_observed_at":"2026-08-15T14:57:30.207852Z","title":"Offlinereinforcementlearning:Tutorial, review, and perspectives on open problems.arXiv preprint arXiv:2005.01643,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.207852Z"},"links":{"cited_paper":"/paper/2005.01643","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:0c44c39c9d1008106ad17fff63fdc495e01ee16571889b515e55260fa0080b1a","observation_id":"de02cbe8-e3ec-48b8-86ef-df1050bf6ff4","resolution":{"observed_at":"2026-08-15T14:57:30.207852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.11027","last_updated":"2023-02-08T09:12:47Z","snapshot_observed_at":"2026-08-18T00:12:07.761554Z","submitted_at":"2022-05-23T04:01:11Z","title":"When Data Geometry Meets Deep Function: Generalizing Offline Reinforcement Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.11027","snapshot_observed_at":"2026-08-15T14:57:30.212100Z","title":"When data geometry meets deep function: Generalizing offline reinforcement learning.arXiv preprint arXiv:2205.11027,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.212100Z"},"links":{"cited_paper":"/paper/2205.11027","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:d110c40d0ac4c83a36d5a68f97fb6e9cc06766a4cbcd6daaf4d64425d9fe9f22","observation_id":"4ce00208-711a-4828-ad89-62c6a668c82e","resolution":{"observed_at":"2026-08-15T14:57:30.212100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.08473","last_updated":"2019-07-06T05:30:14Z","snapshot_observed_at":"2026-08-14T16:44:48.432763Z","submitted_at":"2019-04-17T19:46:02Z","title":"Off-Policy Policy Gradient with State Distribution Correction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.08473","snapshot_observed_at":"2026-08-15T14:57:30.216181Z","title":"Off-policypolicygradientwith state distribution correction.arXiv preprint arXiv:1904.08473,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.216181Z"},"links":{"cited_paper":"/paper/1904.08473","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:b854c56406ca2f7f62278d9f79a72b3e04178d68ec898ca0b1891386b10ad6d3","observation_id":"3e6e6836-309c-49cd-9ff1-285174ff51cb","resolution":{"observed_at":"2026-08-15T14:57:30.216181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.09359","last_updated":"2021-04-24T22:39:30Z","snapshot_observed_at":"2026-08-15T03:52:49.245753Z","submitted_at":"2020-06-16T17:54:41Z","title":"AWAC: Accelerating Online Reinforcement Learning with Offline Datasets","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.09359","snapshot_observed_at":"2026-08-15T14:57:30.224473Z","title":"Awac: Accelerating online reinforcement learning with offline datasets.arXiv preprint arXiv:2006.09359,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.224473Z"},"links":{"cited_paper":"/paper/2006.09359","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:9100dea6c2dadb58360023fd894dca7853d13a60c1f2f7522559e17773f2715a","observation_id":"51416b22-4dde-452d-a45d-59278335e56e","resolution":{"observed_at":"2026-08-15T14:57:30.224473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.00177","last_updated":"2019-10-07T20:23:21Z","snapshot_observed_at":"2026-08-13T13:59:34.315271Z","submitted_at":"2019-10-01T02:23:38Z","title":"Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.00177","snapshot_observed_at":"2026-08-15T14:57:30.228498Z","title":"Advantage-weighted regression: Simple and scalable off-policy reinforcement learning.arXiv preprint arXiv:1910.00177,","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.228498Z"},"links":{"cited_paper":"/paper/1910.00177","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:e9e18f1a5a29eb1f0219eefe98da2d1193ad3ee8c073910d5f9c2fc13f1bbcfc","observation_id":"14b9d6b3-51b8-4049-aaad-e5cc55e07e01","resolution":{"observed_at":"2026-08-15T14:57:30.228498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2105.08140","last_updated":"2021-05-17T20:16:46Z","snapshot_observed_at":"2026-08-16T18:24:08.228597Z","submitted_at":"2021-05-17T20:16:46Z","title":"Uncertainty Weighted Actor-Critic for Offline Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.08140","snapshot_observed_at":"2026-08-15T14:57:30.244330Z","title":"Uncertainty weighted actor-critic for offline reinforcement learning.arXiv preprint arXiv:2105.08140,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.244330Z"},"links":{"cited_paper":"/paper/2105.08140","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:cfd988c46a8d10cec0eb9f73f750b9b700ba127ca39eb6c03c8ce568cd12f47d","observation_id":"685f7411-20ce-4c0a-968f-f571d87c969c","resolution":{"observed_at":"2026-08-15T14:57:30.244330Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.14372","last_updated":"2023-04-19T04:13:38Z","snapshot_observed_at":"2026-08-16T15:52:10.709503Z","submitted_at":"2023-02-28T07:55:02Z","title":"The In-Sample Softmax for Offline Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.14372","snapshot_observed_at":"2026-08-15T14:57:30.248495Z","title":"The in-sample softmax for offline reinforcement learning.arXiv preprint arXiv:2302.14372,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.248495Z"},"links":{"cited_paper":"/paper/2302.14372","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:e9a753f073dbafaef60fd4b8c3fda54c103d3e00b43033b914721485069f9ba2","observation_id":"6265a41e-95b0-4d24-8451-a58bb9a906f0","resolution":{"observed_at":"2026-08-15T14:57:30.248495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.08417","last_updated":"2025-06-10T03:43:22Z","snapshot_observed_at":"2026-08-09T00:05:32.939577Z","submitted_at":"2025-06-10T03:43:22Z","title":"Offline RL with Smooth OOD Generalization in Convex Hull and its Neighborhood","version":1},"cited_work":{"arxiv_id":"2506.08417","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.08417","snapshot_observed_at":"2026-08-15T14:57:30.405591Z","title":"Offline RL with Smooth OOD Generalization in Convex Hull and its Neighborhood","venue":"cs.LG","work_id":"04caac88-37d9-471b-8141-f0403184450a","year":2025},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.257312Z"},"links":{"cited_paper":"/paper/2506.08417","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:3a59629538292c5bcd409cdc6f33a4770d8e052b6c04ca5938b4c9534b04f490","observation_id":"092f5000-38bc-4cb8-a34a-c5af677ff58b","resolution":{"observed_at":"2026-08-15T14:57:30.410209Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:57:30.741141Z","title":"For Gym locomotion, each evaluation uses 10 trajectories, whereas each AntMaze evaluation uses 100 trajectories","venue":null,"work_id":"a47c40a7-400c-43cf-8f6a-5e4e2c15eecf","year":2023},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.261162Z"},"links":{"citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:15ff4df2bf532181626c26121a5f8a2466574ebdcba73496a72f77276aceb7e7","observation_id":"bb8e6db4-684a-4092-97f0-f2ce8480b389","resolution":{"observed_at":"2026-08-15T14:57:30.745195Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"1194.51199","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:57:30.382111Z","title":"The mixture coefficient directly controls the amount of generalized information propagated by bootstrapping","venue":null,"work_id":"45b42425-9f6b-41ad-8707-7c2591903328","year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.265625Z"},"links":{"citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:11c4d37cc96c7aae658eba79cc2ebde9d89e28ac4e228c7b7951caf5e9c3e299","observation_id":"cf67447e-6003-4453-b105-2009f791ce14","resolution":{"observed_at":"2026-08-15T14:57:30.391627Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.20765","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:57:30.543727Z","title":"Less is more: Clustered cross-covariance control for offline rl.arXiv preprint arXiv:2601.20765,","venue":null,"work_id":"5c6f1957-d1de-4c74-a4ca-0f6324232ea7","year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":2001,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.232431Z"},"links":{"citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:6a7ea2285ea2d30428e7e972440a5af263285f6279d457dbe3a603179902e0e8","observation_id":"8f33bbb0-e5d1-4be1-afd5-abcaa9635a06","resolution":{"observed_at":"2026-08-15T14:57:30.549823Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2606.07592","last_updated":"2026-05-28T17:05:05Z","snapshot_observed_at":"2026-07-06T23:47:15.041928Z","submitted_at":"2026-05-28T17:05:05Z","title":"UNIQ: Conformal Calibration for Adaptive Conservatism in Offline Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2606.07592","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.07592","snapshot_observed_at":"2026-08-15T14:57:30.475090Z","title":"UNIQ: Conformal Calibration for Adaptive Conservatism in Offline Reinforcement Learning","venue":"cs.LG","work_id":"e9b9f8e2-5d57-457d-b255-c692a76d7123","year":2026},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.236224Z"},"links":{"cited_paper":"/paper/2606.07592","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:72a0282aa4a97b22448e614fa314abb31af1a4dd3102f44f2c0917da28950025","observation_id":"433cf653-0b74-44f2-9417-53b115e3f9be","resolution":{"observed_at":"2026-08-15T14:57:30.479977Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1911.11361","last_updated":"2019-11-26T06:11:34Z","snapshot_observed_at":"2026-08-12T13:17:36.653110Z","submitted_at":"2019-11-26T06:11:34Z","title":"Behavior Regularized Offline Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.11361","snapshot_observed_at":"2026-08-15T14:57:30.240241Z","title":"Behavior regularized offline reinforcement learning","venue":null,"work_id":null,"year":1911},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.240241Z"},"links":{"cited_paper":"/paper/1911.11361","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:77079e2cf7cdc932e57c0d8f9f7a2b8a9e9d0e027972b5a274ec1be6959d50f9","observation_id":"1ec10dd5-0bad-4ef6-afaa-06772f177aeb","resolution":{"observed_at":"2026-08-15T14:57:30.240241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02328","last_updated":"2023-02-28T22:14:18Z","snapshot_observed_at":"2026-08-17T19:23:07.734442Z","submitted_at":"2023-01-05T23:14:38Z","title":"Extreme Q-Learning: MaxEnt RL without Entropy","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02328","snapshot_observed_at":"2026-08-15T14:57:30.190639Z","title":"Extreme q-learning: Maxent rl without entropy.arXiv preprint arXiv:2301.02328,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.190639Z"},"links":{"cited_paper":"/paper/2301.02328","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:94508a1d7fe31c5a86d43fc2785980a1dde675381f3cab924587b4ad47e10231","observation_id":"804d8ce9-a6bc-43be-a021-0c169128cbf5","resolution":{"observed_at":"2026-08-15T14:57:30.190639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.11097","last_updated":"2022-06-28T09:54:03Z","snapshot_observed_at":"2026-08-17T13:43:22.082666Z","submitted_at":"2021-11-22T10:37:52Z","title":"UMBRELLA: Uncertainty-Aware Model-Based Offline Reinforcement Learning Leveraging Planning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.11097","snapshot_observed_at":"2026-08-15T14:57:30.176724Z","title":"Um- brella: Uncertainty-aware model-based offline reinforcement learning leveraging planning.arXiv preprint arXiv:2111.11097,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.176724Z"},"links":{"cited_paper":"/paper/2111.11097","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:4095b3a50ea0779d39da58b5c8537cb92b5e0c3f3af8c8c465aef9328d57b35a","observation_id":"de025adc-5ed7-48e6-a40c-1d068feefcc9","resolution":{"observed_at":"2026-08-15T14:57:30.176724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2004.07219","last_updated":"2021-02-06T01:57:28Z","snapshot_observed_at":"2026-08-16T08:32:46.407746Z","submitted_at":"2020-04-15T17:18:19Z","title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.07219","snapshot_observed_at":"2026-08-15T14:57:30.181451Z","title":"D4rl: Datasets for deep data-driven reinforcement learning.arXiv preprint arXiv:2004.07219,","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.181451Z"},"links":{"cited_paper":"/paper/2004.07219","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:44d51008cf2f22d12907806cee1325956a7f599aebabc71fc6807eadfb841a53","observation_id":"952896ea-770e-4b55-855f-748acc8ccf9d","resolution":{"observed_at":"2026-08-15T14:57:30.181451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:57:30.172608Z","title":"Flow actor-critic for offline reinforcement learning.arXiv preprint arXiv:2602.18015,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.172608Z"},"links":{"citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:948d9c8da4bd3756d1cdcf7c968e973e59434aef8bc91be946996e4147f01813","observation_id":"fcd56ebf-da73-4c32-a269-bb0cb8890c48","resolution":{"observed_at":"2026-08-15T14:57:30.172608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15810","last_updated":"2023-03-28T08:30:01Z","snapshot_observed_at":"2026-08-16T15:44:47.885182Z","submitted_at":"2023-03-28T08:30:01Z","title":"Offline RL with No OOD Actions: In-Sample Learning via Implicit Value Regularization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.15810","snapshot_observed_at":"2026-08-15T14:57:30.253091Z","title":"arXiv preprint arXiv:2303.15810,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.253091Z"},"links":{"cited_paper":"/paper/2303.15810","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:6093be4f331ce4dc3e776a3ed1bb66c5c36c5c947e09c10e894edf0fcd5eb289","observation_id":"7f5d9b58-8389-43fe-a872-f437c8fd2cd9","resolution":{"observed_at":"2026-08-15T14:57:30.253091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.03647","last_updated":"2020-06-23T16:54:09Z","snapshot_observed_at":"2026-08-15T09:46:30.948536Z","submitted_at":"2020-06-05T19:33:19Z","title":"Deployment-Efficient Reinforcement Learning via Model-Based Offline Optimization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.03647","snapshot_observed_at":"2026-08-15T14:57:30.220449Z","title":"Deployment- efficient reinforcement learning via model-based offline optimization.arXiv preprint arXiv:2006.03647,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:30.220449Z"},"links":{"cited_paper":"/paper/2006.03647","citing_paper":"/paper/2608.03108"},"observation_digest":"sha256:60c8f36646aff17410dacc53170ae6a9bc2699aeb2979baddcc35613a46f8b59","observation_id":"f18bcc3d-6db1-4cda-a9c5-31b83a18726a","resolution":{"observed_at":"2026-08-15T14:57:30.220449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.03108","last_updated":"2026-08-04T04:30:02Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-17T20:37:07.566295Z","submitted_at":"2026-08-04T04:30:02Z","title":"Convex-Hull-Neighborhood Smooth Dual Generalization: Controlling Local Correction Propagation in Offline RL"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":18,"verified_exact":3,"verified_fuzzy":1},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 24 of 24 outbound references and 0 inbound Pith citation observations for arXiv:2608.03108."}