{"as_of":"2026-08-08T11:05:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2bf1211f17a8adeb39d61ade617f383725d1fa469c7d4afce93f9c8c0a587d4c","coverage":[{"denominator":26,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:46:59.972964Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.04730/citation-record","integrity":"/paper/2507.04730/integrity","json":"/paper/2507.04730/citation-record.json","paper":"/paper/2507.04730"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:03.892252Z","title":"Learning Complex Dexterous Manip- ulation with Deep Reinforcement Learning and Demonstrations,","venue":null,"work_id":"cb24c8d3-0caa-4a89-977c-97a370c54e49","year":2018},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:56.890664Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:5c4217e3d7b30e095443000d61193380c30010a0bb15996c88b7bdb506be64d4","observation_id":"7d324a6c-9d4a-45b2-936b-55379f7455ae","resolution":{"observed_at":"2026-08-06T19:47:03.954722Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:03.748967Z","title":"Deep q-learning from demonstrations,","venue":null,"work_id":"ff635c0f-4ff8-4360-804e-2fc4605fe28e","year":2018},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:56.972368Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:9ecb0e2bfa00ff06b7e632aca6a535c926faaf29e1ca832f324700fd3dd687de","observation_id":"275b77ec-66e8-4bb4-8d72-6d13e66314f9","resolution":{"observed_at":"2026-08-06T19:47:03.809877Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:03.596813Z","title":"Action advising with advice imitation in deep reinforcement learning,","venue":null,"work_id":"444d0515-d8c2-4e2a-88fd-52e255404209","year":2021},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:57.065963Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:2eed869df9c1b52d31bf4a8effe31cfc81792af742f993867d7d20fe4e3b7716","observation_id":"2d57bf9a-6545-4fc5-be86-c6979e74f76e","resolution":{"observed_at":"2026-08-06T19:47:03.689134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:03.413729Z","title":"Dqn-tamer: Human-in-the-loop reinforcement learning with intractable feedback,","venue":null,"work_id":"c8acefad-d5aa-4540-a17e-e219b2ce1df7","year":2018},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:57.188486Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:275ea2b127f8191d3f6f0277de5de7b804b0d90973ff9e7da88fa20299354a26","observation_id":"e2361b3e-3fde-43c7-a388-63ab0619c523","resolution":{"observed_at":"2026-08-06T19:47:03.508637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:03.222273Z","title":"Combining manual feedback with sub- sequent mdp reward signals for reinforcement learning","venue":null,"work_id":"0a69b47c-5116-4944-8b01-c65965705a15","year":2010},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:57.290909Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:54507a2ee56a54bf65f0093bf579c17aefb3110d0a6fefbf9ec688ba1f38e0b9","observation_id":"dda975d2-e957-4c35-b55f-5cd2c1ae4252","resolution":{"observed_at":"2026-08-06T19:47:03.329570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.05091","last_updated":"2021-06-09T14:10:50Z","snapshot_observed_at":"2026-08-03T18:30:16.023253Z","submitted_at":"2021-06-09T14:10:50Z","title":"PEBBLE: Feedback-Efficient Interactive Reinforcement Learning via Relabeling Experience and Unsupervised Pre-training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.05091","snapshot_observed_at":"2026-08-06T19:46:57.368857Z","title":"PEBBLE: feedback- efficient interactive reinforcement learning via relabeling experience and unsupervised pre-training,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:57.368857Z"},"links":{"cited_paper":"/paper/2106.05091","citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:004c19ca4ff0226be7224e5eb09efb3c2d0ea20438131fe747ba5b8b60d0b1ab","observation_id":"ee2d89a7-e9a9-4d9b-ab0d-6fb0937dee98","resolution":{"observed_at":"2026-08-06T19:46:57.368857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:02.990790Z","title":"Interactive learning with corrective feedback for policies based on deep neural networks,","venue":null,"work_id":"28d03703-d103-4649-814a-dee9700c29b3","year":2018},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:57.460776Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:c0a290648acfd7a87b2cc345e5797fa010900497043c7b57c2280349c0e2db0a","observation_id":"79ff446e-8da7-4570-8fae-dcbd72a0333b","resolution":{"observed_at":"2026-08-06T19:47:03.113145Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:02.840666Z","title":"Reinforcement learning of motor skills using policy search and human corrective advice,","venue":null,"work_id":"fac5c4bb-acd5-470b-b4e1-0e246ff3e995","year":2019},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:57.539086Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:3baab93230c9734855161fa8c56fe3850cc4c1bfb4232c54a29dc02be930e009","observation_id":"4e9d4548-a904-42cf-901c-2dbf6c0cd36a","resolution":{"observed_at":"2026-08-06T19:47:02.916151Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:02.619671Z","title":"No, to the right: Online language corrections for robotic manipulation via shared autonomy,","venue":null,"work_id":"43d7a6c8-e159-4e76-ba48-498a565bda23","year":2023},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:57.686163Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:de5ef0e8c41d9cd6aa1aed9e4f67e8d851313d68fa93a5bd4c06787822140622","observation_id":"82101eaf-4938-4dc9-8a2a-bec3b2fdefd3","resolution":{"observed_at":"2026-08-06T19:47:02.701103Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12910","last_updated":"2024-03-19T17:08:24Z","snapshot_observed_at":"2026-08-04T09:24:08.528569Z","submitted_at":"2024-03-19T17:08:24Z","title":"Yell At Your Robot: Improving On-the-Fly from Language Corrections","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12910","snapshot_observed_at":"2026-08-06T19:46:57.823707Z","title":"Yell at your robot: Improving on-the-fly from language corrections,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:57.823707Z"},"links":{"cited_paper":"/paper/2403.12910","citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:2267c26449ab8d179f5629973e782b1319229e81ebfae9dd1eb1d6166aaf4d4c","observation_id":"147de395-621b-457c-a72f-cab7571b85ec","resolution":{"observed_at":"2026-08-06T19:46:57.823707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/s10846-018-0839-z","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:00.112058Z","title":"An interactive framework for learning continuous actions policies based on corrective feedback,","venue":null,"work_id":"8654bf94-ae53-4b0f-a8a2-082ccc187454","year":2019},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:57.999108Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:db05d8986819e4b43d6c7d7bbb54119da14dcbe303954dbb879689c26f8cbdad","observation_id":"4283aa42-c4de-43e0-b529-6e8cb74fc9f8","resolution":{"observed_at":"2026-08-06T19:47:00.199565Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:02.412657Z","title":"Algorithms for inverse reinforcement learning,","venue":null,"work_id":"13f9cd96-6aab-487b-9dd1-f1c9965ad1be","year":2000},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:58.116908Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:aeb495fac10ee5f35eece5250dda284b55f2e7f37a1a3b0cfcea7b053bcdc576","observation_id":"935d72a3-3053-488f-8139-1c70e4f5a339","resolution":{"observed_at":"2026-08-06T19:47:02.503736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:02.221186Z","title":"Interactively shaping agents via human reinforcement: The TAMER framework,","venue":null,"work_id":"49b671f0-4670-47f8-8b6f-05d158123316","year":2009},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:58.292369Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:0f6be9cc0effc33a1d6253a6255a08e1da495772206f768d1da139609082858c","observation_id":"6d80a485-d838-4d3f-821e-7b0a46936b8b","resolution":{"observed_at":"2026-08-06T19:47:02.326424Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:01.988737Z","title":"Interactive learning from policy-dependent human feedback,","venue":null,"work_id":"f249cd53-e37e-485d-89d9-ee1f7411ec99","year":2017},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:58.448656Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:43ebe90d6b5160fc3c70d283009ff1dd92ec8d6b4f6a20654707fbf20c8b2cae","observation_id":"ff3522ad-c95c-4bdf-8712-fad249077b2e","resolution":{"observed_at":"2026-08-06T19:47:02.079797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:01.749332Z","title":"Deep reinforcement learning from human preferences,","venue":null,"work_id":"6f8c8d94-bff1-4092-86db-e99676535b16","year":2017},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:58.646118Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:859b6892f3730e559ce9acd9f12561ffe043e790113727fdbcd8e03c94a5257c","observation_id":"5f5a78cb-696a-4f5e-af77-b26a72e8fb48","resolution":{"observed_at":"2026-08-06T19:47:01.823057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:01.540709Z","title":"Few-shot preference learning for human- in-the-loop rl,","venue":null,"work_id":"53654533-83c2-4361-ada3-1014a8986291","year":2022},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:58.820859Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:3476e6078fb4e6ab7aa95053884a009bf9aacd942eac09a1926ee6d8402dc2aa","observation_id":"444caa52-0700-4b11-bc43-9d1ed8f5799b","resolution":{"observed_at":"2026-08-06T19:47:01.663685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:01.331407Z","title":"Integrating behavior cloning and reinforcement learning for improved performance in sparse reward environments,","venue":null,"work_id":"321763bf-5e07-491c-8ef9-bb100349e694","year":2019},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:58.938867Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:0bec7c832be842ab1aa45be8c90dccac14af7c8bcbd986ab64ae1553f9ce20fa","observation_id":"ff089631-31ee-44d3-b19f-88e63eda4736","resolution":{"observed_at":"2026-08-06T19:47:01.411300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:01.117504Z","title":"Agent- advising approaches in an interactive reinforcement learning scenario,","venue":null,"work_id":"129b3e03-2ea4-4842-aa52-3eaf8f6d5c3e","year":2017},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:59.082151Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:00061a190d74f22c800275d88f8132afffb590776c76fd07bddd842311516218","observation_id":"c843d469-34c9-46df-b213-c841243fe7fa","resolution":{"observed_at":"2026-08-06T19:47:01.218622Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:00.945516Z","title":"Modem: Accelerating visual model-based reinforcement learning with demonstrations,","venue":null,"work_id":"d25948c0-be60-43be-9e2f-d36a513b7c3d","year":2022},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:59.166682Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:4b160daf85e3daaacc688da2e586bd5177f1807fa9316dbb2e6b43b6b4084a99","observation_id":"a0e479ed-1cf6-425e-8774-279f96601423","resolution":{"observed_at":"2026-08-06T19:47:01.035675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:46:59.268947Z","title":"Interactive robot learning from verbal correction,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:59.268947Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:083175d2a980077e2be749ff654b7fb41994e1eac9372fd6dfa99b762e014eda","observation_id":"00d80b20-ee01-4682-8da9-061fd5a1db2f","resolution":{"observed_at":"2026-08-06T19:46:59.268947Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:00.705515Z","title":"A reduction of imitation learning and structured prediction to no-regret online learning,","venue":null,"work_id":"7f67d2f0-4305-4cd6-aee5-61cf34deb679","year":2011},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:59.375451Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:7527a5edb627a03242e2f877a59ec8424f14138ed82c40585f9203c740b15b40","observation_id":"367dae9a-9353-44d3-8ad8-533c927286e6","resolution":{"observed_at":"2026-08-06T19:47:00.801971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:46:59.507262Z","title":"Orbit: A unified simulation framework for interactive robot learning environments,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:59.507262Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:fbbc686906964346a64c9f47c301f4f95310e122bb091554ea7eb58efbcc2c0e","observation_id":"f182bb5a-b9af-4189-93f1-5341d9f71a8b","resolution":{"observed_at":"2026-08-06T19:46:59.507262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:00.482668Z","title":"Anymal - a highly mobile and dynamic quadrupedal robot,","venue":null,"work_id":"8a10777a-fcdf-4804-93d0-1f0a72f5e720","year":2016},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:59.632097Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:5939da7b9f90f0f792c38eeb0fcdb1ee021cc66acf508f8b2bfef17472d500d5","observation_id":"441a8885-0384-41ee-98cf-3ecda6a3a4fc","resolution":{"observed_at":"2026-08-06T19:47:00.562384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:46:59.739184Z","title":"Deep reinforcement learning with double q-learning,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:59.739184Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:54110b137c4434bcc80754a2c757581c6042a366a1e96234fa2a3b5daa9e9ebe","observation_id":"1de215f1-4210-429d-8fea-5d79092ad54e","resolution":{"observed_at":"2026-08-06T19:46:59.739184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:46:59.839291Z","title":"Probabilistic roadmaps for path planning in high-dimensional configuration spaces,","venue":null,"work_id":null,"year":1996},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:59.839291Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:50cdf367a0eeead8141a11e92f1b654b7c4b0cea444d3a4acb3af82f3e7d81df","observation_id":"ba101753-c005-411e-ad8a-d94ef21d471f","resolution":{"observed_at":"2026-08-06T19:46:59.839291Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:47:00.316327Z","title":"Reinforcement learning: An introduction,","venue":null,"work_id":"52b3265f-6005-4f3a-b23b-01c231b2050d","year":1998},"citing_paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T19:46:59.972964Z"},"links":{"citing_paper":"/paper/2507.04730"},"observation_digest":"sha256:f0512ccc8395938c0e70c30908ffba4e1c75befda4c8b55483f4725f3d872376","observation_id":"af2ff169-3898-4f90-a720-2dc33086bd1c","resolution":{"observed_at":"2026-08-06T19:47:00.386268Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.04730","last_updated":"2025-07-07T07:54:28Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-06T19:38:28.411769Z","submitted_at":"2025-07-07T07:54:28Z","title":"CueLearner: Bootstrapping and local policy adaptation from relative feedback"},"reference_resolution":{"displayed":26,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":6,"verified_exact":1,"verified_fuzzy":19},"total_outbound_references":26},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 26 of 26 outbound references and 0 inbound Pith citation observations for arXiv:2507.04730."}