{"as_of":"2026-08-10T05:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:849d85ef7b8bf9e5ce8dea811e6e8a37d0b083d9656cd4b0d0eabcb4fe74406e","coverage":[{"denominator":63,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":63,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T05:02:13.369371Z","state":"measured"},{"denominator":63,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":63,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2502.03369/citation-record","integrity":"/paper/2502.03369/integrity","json":"/paper/2502.03369/citation-record.json","paper":"/paper/2502.03369"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1701.04079","last_updated":"2017-01-15T17:14:40Z","snapshot_observed_at":"2026-08-09T21:22:39.238224Z","submitted_at":"2017-01-15T17:14:40Z","title":"Agent-Agnostic Human-in-the-Loop Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1701.04079","snapshot_observed_at":"2026-08-09T05:02:13.097061Z","title":"Agent-agnostic human-in-the-loop reinforcement learning","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.097061Z"},"links":{"cited_paper":"/paper/1701.04079","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:186c495cd00e6c97aeb4729669baefe932a960fb438f3db0a35d5665015c6cd8","observation_id":"7d390698-79ce-46a3-a28b-e79ecaabfe62","resolution":{"observed_at":"2026-08-09T05:02:13.097061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:14.038552Z","title":"Constrained policy optimization","venue":null,"work_id":"0de543b9-ae33-4d21-a785-22a9d8f8c590","year":2017},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.102328Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:1e12b224b5c029dc47143e9f5e663c4a5562a014bf0411f281a0e0f3865334b9","observation_id":"1f779148-5b50-4f46-ab15-ea9f0849d2da","resolution":{"observed_at":"2026-08-09T05:02:14.042162Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:14.027729Z","title":"An interactive framework for learning continuous actions policies based on corrective feedback","venue":null,"work_id":"0629ccd0-1cd3-4526-8f9b-c62f70deeacb","year":2019},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.106908Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:346bfa26cbe6ee76872091010d11bd61efe3086dc2467b05b35ec3c14bb488c7","observation_id":"6551bf87-7e4c-4241-9215-45c03fb950f1","resolution":{"observed_at":"2026-08-09T05:02:14.031627Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:14.016665Z","title":"Minimalistic gridworld environment for openai gym","venue":null,"work_id":"14755568-e42a-465e-abee-7b3a02cea9c8","year":2018},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.111768Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:cba831bbb77f3874d667469cdbb4e122f476a8594211f8ef5faa0f889526bc4e","observation_id":"9f113462-bbb9-4ff1-bc2c-4d1475b62433","resolution":{"observed_at":"2026-08-09T05:02:14.020818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.115607Z","title":"Christiano, Jan Leike, Tom B","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.115607Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:85412e670ee787dfbc296e786f73a0c0afb2f871da084523007571422b71f6cf","observation_id":"43d96413-8c44-46c5-a6ed-b9ad7354217f","resolution":{"observed_at":"2026-08-09T05:02:13.115607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.08630","last_updated":"2020-12-15T21:39:50Z","snapshot_observed_at":"2026-08-09T08:09:12.994031Z","submitted_at":"2020-12-15T21:39:50Z","title":"Open Problems in Cooperative AI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.08630","snapshot_observed_at":"2026-08-09T05:02:13.119853Z","title":"Open problems in cooperative ai","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.119853Z"},"links":{"cited_paper":"/paper/2012.08630","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:8ee46478946e2a7e8224f1384643a5ab5c51c40ca36a453435d46c5b1ccd4172","observation_id":"a61c1ff7-38cd-4a67-a958-2eb43de93f67","resolution":{"observed_at":"2026-08-09T05:02:13.119853Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.999490Z","title":"Magnetic control of tokamak plasmas through deep reinforcement learning","venue":null,"work_id":"d0b31dcf-77d9-404a-95ac-169a2695e257","year":2022},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.125143Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:a250a555922c33b1793a0fdc745af267a0243fc0b86f5127074073cc99e34773","observation_id":"4ef4784d-5dbf-4407-9428-04d38b84f334","resolution":{"observed_at":"2026-08-09T05:02:14.003269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.129124Z","title":"CARLA: An open urban driving simulator","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.129124Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:6356d39baf0436b5763c971b17ff541beb89fb3c4c10ea09297b437c8ba14016","observation_id":"5cbdc2e8-a9e3-4a6e-92db-05407621c93c","resolution":{"observed_at":"2026-08-09T05:02:13.129124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.982126Z","title":"Learning robust rewards with adverserial inverse reinforcement learning","venue":null,"work_id":"04766299-4cd1-41f8-8683-88d0f5e3b02c","year":2018},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.132885Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:09237f2198658d3768689923d02250a6d98ce5a49f86d3e59841ec709a3fc446","observation_id":"e5614c5d-7d61-4601-9830-9e1c162b2a6f","resolution":{"observed_at":"2026-08-09T05:02:13.985809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.971862Z","title":"Addressing function approximation error in actor-critic methods","venue":null,"work_id":"146c25e5-b170-453e-a883-c25ba57ecf59","year":2018},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.136365Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:e99bec5ebc056479f61964de3806236efcb46229b578effc5149c483a75fd497","observation_id":"7b39b107-02c1-44ed-804f-de83d3e30c30","resolution":{"observed_at":"2026-08-09T05:02:13.975624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.960779Z","title":"Widening the pipeline in human-guided reinforcement learning with explanation and context-aware data augmentation","venue":null,"work_id":"5894c508-0f00-4c1f-8512-ea7617cb80f2","year":2021},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.140422Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:ade513c6b973e56bb119d5682fd3816f64a58705a2f7b66200b7514a0b3c5014","observation_id":"7222c1b5-b6b1-4e9e-a42b-73c4e1fd9997","resolution":{"observed_at":"2026-08-09T05:02:13.964698Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.950215Z","title":"Learning to walk in the real world with minimal human effort, 2020","venue":null,"work_id":"80c2ff27-5fbb-4522-a5c4-d69274ae973b","year":2020},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.143910Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:8b77657081441a2a9e7724735515061dfe3185c27afd8254c20bde1f4dd80285","observation_id":"8aac2172-d86a-40e0-b6c3-1092cba33595","resolution":{"observed_at":"2026-08-09T05:02:13.953813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.939764Z","title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","venue":null,"work_id":"ca010d99-162b-4df6-a0d5-3b05c76e5a0e","year":2018},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.147681Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:dfb47db19a57f9c53c3003d27126cf36fda2af86d833ee7ee9554f13f0b172dd","observation_id":"5814ac2b-f61b-4abc-a940-19b105a23d71","resolution":{"observed_at":"2026-08-09T05:02:13.943700Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.928558Z","title":"Generative adversarial imitation learning","venue":null,"work_id":"ba63921d-30a8-46ee-a74d-4059ac6def84","year":2016},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.150972Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:5b93f2eedcb78b4cefd261ffdb42f7bf3b83c7902c5ead34d2bf2b67e1bf4d0c","observation_id":"a12310a7-8404-448e-b892-16650f1dd590","resolution":{"observed_at":"2026-08-09T05:02:13.932348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.00160","last_updated":"2019-05-07T02:55:37Z","snapshot_observed_at":"2026-08-08T23:53:27.179051Z","submitted_at":"2019-05-01T02:14:22Z","title":"Precise Synthetic Image and LiDAR (PreSIL) Dataset for Autonomous Vehicle Perception","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.00160","snapshot_observed_at":"2026-08-09T05:02:13.156609Z","title":"Waslander","venue":null,"work_id":null,"year":1905},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.156609Z"},"links":{"cited_paper":"/paper/1905.00160","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:29c9d11d225c7307291711cac7f539b3c91f05f1996a0c1d3e671904e8b36ba8","observation_id":"2d9f4182-d0a1-46b2-a062-953718f16410","resolution":{"observed_at":"2026-08-09T05:02:13.156609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.918280Z","title":"Learning to share autonomy across repeated interaction","venue":null,"work_id":"5cd665d7-4168-46eb-8449-9d877af0338c","year":2021},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.175680Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:7e73707258351d33782374aef608da9308b775bac23a2b8bda69be04ad9dd4c0","observation_id":"2d88299e-4df6-4c4a-8b8b-cf1c3d882959","resolution":{"observed_at":"2026-08-09T05:02:13.921905Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.907708Z","title":"Hg-dagger: Interactive imitation learning with human experts","venue":null,"work_id":"46042894-3a59-44f2-ae04-81e4308f6bae","year":2019},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.195567Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:effbcd79ea43812c58bdb767c66c822d05dcaaff16499167db4327e6a22f5e20","observation_id":"bbbdf038-5ea7-40bf-b2db-ddc029307981","resolution":{"observed_at":"2026-08-09T05:02:13.911648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.897888Z","title":"Learning to drive in a day","venue":null,"work_id":"f0769e85-c6b7-4bc4-aa10-faf4f985f216","year":2019},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.199268Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:e1a6212f0b7810e677b1209e4b7e9ef64c173a5da650134e4040ebd427e655a9","observation_id":"515442c5-2d78-450d-8dab-92b0fa12c494","resolution":{"observed_at":"2026-08-09T05:02:13.901302Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.887906Z","title":"Reinforcement learning from human reward: Discounting in episodic tasks","venue":null,"work_id":"70c43a53-7b76-4698-9c1e-0882f1df095e","year":2012},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.202906Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:3396bfa6cb08831d0129de53ddaf55dbcb13dab6d3293edff6bd35a6539efb35","observation_id":"12e15a05-11a0-42cc-a893-0fdd3527d49e","resolution":{"observed_at":"2026-08-09T05:02:13.891483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.877348Z","title":"Specification gaming: the flip side of ai ingenuity","venue":null,"work_id":"3ad1f71b-ac2b-4429-a616-8116e5675c7b","year":2020},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.206101Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:281f8a00421f0c57c72d4199dd387de8e40668da2dcd610afb968b506668555e","observation_id":"8ec8c7b7-888c-4308-a845-7c7bf919bb68","resolution":{"observed_at":"2026-08-09T05:02:13.880895Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.867113Z","title":"Conservative q-learning for offline reinforcement learning","venue":null,"work_id":"131900d6-08f5-43bc-a36e-ac7d35d818d4","year":2020},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.209424Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:c472b454cae116686eb7094d6bc3c11e2c6a3261345c4367505d6a0038015ba9","observation_id":"eae9d699-d97b-4fa8-8a8f-b935d18a6caf","resolution":{"observed_at":"2026-08-09T05:02:13.870747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.05091","last_updated":"2021-06-09T14:10:50Z","snapshot_observed_at":"2026-08-09T12:10:10.862824Z","submitted_at":"2021-06-09T14:10:50Z","title":"PEBBLE: Feedback-Efficient Interactive Reinforcement Learning via Relabeling Experience and Unsupervised Pre-training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.05091","snapshot_observed_at":"2026-08-09T05:02:13.213302Z","title":"Pebble: Feedback-efficient interactive reinforcement learning via relabeling experience and unsupervised pre-training","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.213302Z"},"links":{"cited_paper":"/paper/2106.05091","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:9a63157287a16577f429859f0a0e54f7ea39972df6bc8f4381be8a7911326dac","observation_id":"480fa63f-cede-40c0-b88d-6a38f8e293db","resolution":{"observed_at":"2026-08-09T05:02:13.213302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1811.07871","last_updated":"2018-11-19T18:48:04Z","snapshot_observed_at":"2026-07-06T07:15:51.575438Z","submitted_at":"2018-11-19T18:48:04Z","title":"Scalable agent alignment via reward modeling: a research direction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1811.07871","snapshot_observed_at":"2026-08-09T05:02:13.217444Z","title":"Scalable agent alignment via reward modeling: a research direction","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.217444Z"},"links":{"cited_paper":"/paper/1811.07871","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:f426f205558abf148e2ead7d6f05ea0626279848dc11353f3a85a561b0117233","observation_id":"ab5c7713-fe20-4ef9-a9f2-ad7da4b506d9","resolution":{"observed_at":"2026-08-09T05:02:13.217444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.01643","last_updated":"2020-11-01T23:50:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-04T17:00:15Z","title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.01643","snapshot_observed_at":"2026-08-09T05:02:13.221593Z","title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.221593Z"},"links":{"cited_paper":"/paper/2005.01643","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:c1a04d8b7ea6c7e86abd16b6ee8f6b1701a1748a18124a4ea607caa3c03c6679","observation_id":"800cc48d-9186-41e3-bec0-9b126f665946","resolution":{"observed_at":"2026-08-09T05:02:13.221593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.856440Z","title":"Metadrive: Composing diverse driving scenarios for generalizable reinforcement learning","venue":null,"work_id":"1ea46f31-5427-4fbf-a5fd-dd78d4f4b9a7","year":2022},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.225792Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:2eb11a5f9e54adbb97b4e16e708463bf8c94f7c464369c424dfdb621e875f06f","observation_id":"41784e27-ff14-4fe3-a9ec-8d65cb273d50","resolution":{"observed_at":"2026-08-09T05:02:13.860131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.846048Z","title":"Efficient learning of safe driving policy via human-ai copilot optimization","venue":null,"work_id":"af6c9308-9dea-4e67-9afb-57ddcb942a91","year":2022},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.229681Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:ec8ea98e72d32755fd2bf0bd38cdd44d118935a62cfec462c1538426d5cca852","observation_id":"613635c1-2af8-4d10-9e08-ba39b93faf50","resolution":{"observed_at":"2026-08-09T05:02:13.849745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.835520Z","title":"Interactive learning from policy-dependent human feedback","venue":null,"work_id":"ce2d12c0-f76a-409f-94e4-e92dadf495b0","year":2017},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.233935Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:12ac83b1169929ba4fb4ec53fe0b79e55ce90da8bc2a197119b027021d69fc10","observation_id":"6b5d3aa1-10ac-46a2-adf8-93d386449437","resolution":{"observed_at":"2026-08-09T05:02:13.839140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.824987Z","title":"Where to add actions in human-in- the-loop reinforcement learning","venue":null,"work_id":"656ef0ee-0bbe-4122-b119-5855221b1667","year":2017},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.237652Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:02b9430a44a7e2297602188d6cf05e6e750083bb7ccf9f319375ae7ae8254197","observation_id":"c17b998b-b5a3-4ca8-bef6-efdf4058b2bf","resolution":{"observed_at":"2026-08-09T05:02:13.829020Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.06733","last_updated":"2020-12-12T05:30:35Z","snapshot_observed_at":"2026-08-09T12:35:25.982729Z","submitted_at":"2020-12-12T05:30:35Z","title":"Human-in-the-Loop Imitation Learning using Remote Teleoperation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.06733","snapshot_observed_at":"2026-08-09T05:02:13.241309Z","title":"Human- in-the-loop imitation learning using remote teleoperation","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.241309Z"},"links":{"cited_paper":"/paper/2012.06733","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:085d7b01fee067f3c629dd0204e1f3d16a0ed8b0b0ee2d28b56fb2bc858fbe39","observation_id":"1010a489-38aa-4c5b-b1b4-c5e7c20ef80b","resolution":{"observed_at":"2026-08-09T05:02:13.241309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.814829Z","title":"Ensembledagger: A bayesian approach to safe imitation learning","venue":null,"work_id":"0b50c199-0d69-4207-8053-ee8f6a60e496","year":2019},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.245371Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:6a18c20ef9943b311dec7fb5555b42b29109238bdb48ab866ae38ac6bf6b71fd","observation_id":"60815919-ae78-415b-9ebc-0f7f148902dc","resolution":{"observed_at":"2026-08-09T05:02:13.818504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.249898Z","title":"Human-level control through deep reinforcement learning","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.249898Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:69d74b40898f1150bf1a82de2fe90f6b3f33264b4005d3300d8674654975b54a","observation_id":"ced1ce6b-ef94-4b41-b8bb-c876ef40c9b6","resolution":{"observed_at":"2026-08-09T05:02:13.249898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.798817Z","title":"Interactively shaping robot behaviour with unlabeled human instructions","venue":null,"work_id":"b535946f-4819-449d-bfa2-10b2819d1be2","year":2020},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.253953Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:3e1683ff16b5aaaabee06df47a892a5685d984a51e45d468140f68db8e13f634","observation_id":"5e3bba32-8576-4491-8fe4-a9c3e45548cb","resolution":{"observed_at":"2026-08-09T05:02:13.802496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.788615Z","title":"Deep exploration via bootstrapped DQN","venue":null,"work_id":"f6e4e69c-a756-4501-971b-653f81bf0531","year":2016},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.257745Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:73bc4e4a426f638a6d9c74a1f569fd4b5a374ef8517f958c0355829f45b9b42b","observation_id":"485f51ee-91ce-4237-afda-1ca95d0b9bde","resolution":{"observed_at":"2026-08-09T05:02:13.792031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-08-09T05:02:13.261475Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.261475Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:8483047dfc6ae25d25a55d172d82f4392952b601a49c30053d7ca92a6c5e4559","observation_id":"ee7e1128-c1dd-45f1-be1b-0f85cc47d448","resolution":{"observed_at":"2026-08-09T05:02:13.261475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.778366Z","title":"Deeptake: Prediction of driver takeover behavior using multimodal data","venue":null,"work_id":"e3f9dddb-1efc-4559-a2b4-d6abb2d83714","year":2021},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.266006Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:f940004692ec7a266eb3c766a19559e7e3d057dda4a04adb8c9ec4493f3cd947","observation_id":"2f37cc8a-19ae-4da6-b640-437947344033","resolution":{"observed_at":"2026-08-09T05:02:13.782039Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.768378Z","title":"Learning reward functions by integrating human demonstrations and preferences","venue":null,"work_id":"d67205f8-4e63-4ede-ad0a-cb18d615af69","year":2019},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.269625Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:9e10dfd0ed5cbd38c2f203ab8570480423f1a4a030c6f8387a87c7aa27584005","observation_id":"d77a5421-8af5-435e-a7f1-2d5400274dd5","resolution":{"observed_at":"2026-08-09T05:02:13.771865Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.758040Z","title":"Stable-baselines3: Reliable reinforcement learning implementations","venue":null,"work_id":"26738627-2fc1-473d-a0ef-2336a6534700","year":2021},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.273446Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:1e41fd02300211c6d3eb2ece59ec2c226b8ff0a5ed1c9dd2d535cbe2a4bf746a","observation_id":"3504d69c-1fdf-4a9a-b3b2-fb81bd3064b2","resolution":{"observed_at":"2026-08-09T05:02:13.761852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.747501Z","title":"Shared autonomy via deep reinforcement learning","venue":null,"work_id":"cb257ebd-f19c-4183-aff0-c6f7fb662351","year":2018},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.277130Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:32237a2a551ed36edc15cf82ba612d161223d076455878eab6e701df8a71342a","observation_id":"a4aa344c-5a33-4a04-b2c4-819b55859e9f","resolution":{"observed_at":"2026-08-09T05:02:13.751260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.737394Z","title":"Efficient reductions for imitation learning","venue":null,"work_id":"9e4350d4-20f9-4d62-9dee-df1e95d0556a","year":2010},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.280788Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:c42e6ab257a3784ebde36ff6927ae2fb8cb2815510d775c5fd22aff4212b41e9","observation_id":"2a7b02cd-c928-4ab8-8a6f-e684a891d10e","resolution":{"observed_at":"2026-08-09T05:02:13.741060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.727277Z","title":"Human compatible: Artificial intelligence and the problem of control","venue":null,"work_id":"eb1cd30d-0933-4300-8a0d-af832634aa89","year":2019},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.284438Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:4d549e8f819d505178824d14ed8c82bc2aa6b053fdd8d3ba092084044210583c","observation_id":"512fd1f9-fd5d-493b-8cf5-ce44b21155d6","resolution":{"observed_at":"2026-08-09T05:02:13.730757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.716690Z","title":"Active preference-based learning of reward functions","venue":null,"work_id":"84719c5a-924d-4a35-bde9-33516c422a3d","year":2017},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.287996Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:73c1b04af7f2e9ed64d9de30a6c41a72fe2d278d6c035f04530b585f4e3be0ab","observation_id":"ef5d8521-5e86-45ba-8c00-014e4e319f1a","resolution":{"observed_at":"2026-08-09T05:02:13.720405Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1902.04043","last_updated":"2019-12-09T07:26:52Z","snapshot_observed_at":"2026-08-10T01:55:31.555267Z","submitted_at":"2019-02-11T18:43:53Z","title":"The StarCraft Multi-Agent Challenge","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1902.04043","snapshot_observed_at":"2026-08-09T05:02:13.291676Z","title":"The starcraft multi-agent challenge","venue":null,"work_id":null,"year":1902},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.291676Z"},"links":{"cited_paper":"/paper/1902.04043","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:052b02fc2127a9f4bbf194793f2f161edecfe21d5bfa416b0680d1cebb8047c1","observation_id":"43b7a1ab-eb22-4dd8-92f6-0e5c6ad147e6","resolution":{"observed_at":"2026-08-09T05:02:13.291676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.706875Z","title":"Trial without error: Towards safe reinforcement learning via human intervention","venue":null,"work_id":"0026f108-ccbf-40ad-861f-bc34e234a1b9","year":2018},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.296095Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:d11ca7d6b57ec7f91072c1ca1295e8439503a513a56386034cd222101a137fd9","observation_id":"b959a36c-801b-41d3-becb-2a8a2b0779bf","resolution":{"observed_at":"2026-08-09T05:02:13.710352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1511.05952","last_updated":"2016-02-25T17:55:31Z","snapshot_observed_at":"2026-07-06T04:37:00.543211Z","submitted_at":"2015-11-18T20:54:44Z","title":"Prioritized Experience Replay","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1511.05952","snapshot_observed_at":"2026-08-09T05:02:13.299795Z","title":"Prioritized experience replay","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.299795Z"},"links":{"cited_paper":"/paper/1511.05952","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:eb640bf4a5e436b9adbd30c9142db019d625e41a49011f4dc98f88cb7af8dc16","observation_id":"a52999c7-3485-4783-b7a0-55bf2061e842","resolution":{"observed_at":"2026-08-09T05:02:13.299795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-09T05:02:13.303556Z","title":"Proximal policy optimization algorithms","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.303556Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:e5954aa62c80754032be2321502b9fede380b8cda80a14cb9d0f666d0c27c558","observation_id":"94d15502-5157-47ab-8d20-976a467ab046","resolution":{"observed_at":"2026-08-09T05:02:13.303556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.306927Z","title":"Mastering the game of go with deep neural networks and tree search","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.306927Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:bef1d8884bd3518f127b7a33cb9e0ecb70ecb66024b24a47fd788d96b471aa25","observation_id":"0398c6e5-90a9-4f6b-a123-049b82a417d6","resolution":{"observed_at":"2026-08-09T05:02:13.306927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.691485Z","title":"Learning from interventions","venue":null,"work_id":"a4a513ee-5a72-487f-ba11-fa51cb512fbb","year":2020},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.310183Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:18eef9627c0f678ffe0609e43eabdf23e8d004ff0e0dd90fe06260dc1e9534d1","observation_id":"7fd6d2e5-f549-4dab-a89d-eb87b501cb25","resolution":{"observed_at":"2026-08-09T05:02:13.694734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.681831Z","title":"Responsive safety in reinforcement learning by PID lagrangian methods","venue":null,"work_id":"c00ff4db-4850-4a2d-93c2-4f466ca24062","year":2020},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.313523Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:790812704c2449a8b9578ff41f484791f2024c3364c5f5cadc4663b37e94de9a","observation_id":"c3b36b0b-6a65-4bec-bebb-593a2260a381","resolution":{"observed_at":"2026-08-09T05:02:13.685480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.672077Z","title":"Intervention aided reinforcement learning for safe and practical policy optimization in navigation","venue":null,"work_id":"f433dd61-6c73-44e1-b4d6-38be93675b14","year":2018},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.316900Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:d6275452c00a1aa92ed04102ffaa96c4c85d26e9ef56374924214435b3d76356","observation_id":"16f063e0-5ee7-4fe0-a36f-97e2e770258a","resolution":{"observed_at":"2026-08-09T05:02:13.675802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.661661Z","title":"Appli: Adaptive planner parameter learning from interventions","venue":null,"work_id":"40538cb5-29ab-45b2-b369-0fa819e41c04","year":2021},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.320318Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:90b09bd6865cb8d46f4c15bd749e6279b6d9f9b89a2a82b640cb81a1a5754071","observation_id":"1f9894bd-0bf9-4556-8a26-6636f5946c53","resolution":{"observed_at":"2026-08-09T05:02:13.665765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.651611Z","title":"Apple: Adaptive planner parameter learning from evaluative feedback","venue":null,"work_id":"98713356-d966-4ec8-9699-9b0b0d12e1b5","year":2021},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.323639Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:d11ad075e79767b96bbfba6bd91e3214d984a60effdae2be8d7e6539131fab31","observation_id":"478fe29a-200e-4244-aaec-9c1429f8ed03","resolution":{"observed_at":"2026-08-09T05:02:13.655365Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.641354Z","title":"Waytowich, Vernon Lawhern, and Peter Stone","venue":null,"work_id":"53780101-0ab1-4651-9484-677d7087b1f0","year":2018},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.326895Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:88569412520bcabd061747bed7b38892fdc302f335e4a0ee9649d8da4cb58356","observation_id":"f6731b1a-369e-46d1-ab34-e27019e4ca7c","resolution":{"observed_at":"2026-08-09T05:02:13.644940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.631183Z","title":"A survey of preference-based reinforcement learning methods","venue":null,"work_id":"1664aab8-93ed-485c-845c-b9f213d649e7","year":2017},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.330249Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:2938cb3257753bfe5dc34172d6b27382133296f691d56a81992d5af4681aac7c","observation_id":"9efe55a1-6684-4c74-95cc-3541b6f622e2","resolution":{"observed_at":"2026-08-09T05:02:13.635146Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.620733Z","title":"Look before you leap: Safe model-based reinforcement learning with human intervention","venue":null,"work_id":"89002bf7-4869-4d6c-9049-d63356aa7f9b","year":2022},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.333912Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:bb20f9bb476a3e06db44fe63531dfc9b16d5ddd1ac858cfa6fb271139039cdd7","observation_id":"fff5f2f2-54f5-4012-8fb2-f5dd6ee30d87","resolution":{"observed_at":"2026-08-09T05:02:13.624382Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.01741","last_updated":"2022-07-08T16:22:28Z","snapshot_observed_at":"2026-08-10T00:32:47.173623Z","submitted_at":"2022-02-03T18:04:54Z","title":"How to Leverage Unlabeled Data in Offline Reinforcement Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.01741","snapshot_observed_at":"2026-08-09T05:02:13.337362Z","title":"How to leverage unlabeled data in offline reinforcement learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.337362Z"},"links":{"cited_paper":"/paper/2202.01741","citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:aee51a44651b21a865727d05918dbd8241a2a0b5d4fd3c159271bd9dddb02d99","observation_id":"4a6b2247-a7b6-4c53-8a33-62eec05ac09a","resolution":{"observed_at":"2026-08-09T05:02:13.337362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.609724Z","title":"Query-efficient imitation learning for end-to-end simulated driving","venue":null,"work_id":"8e4e0587-9976-49e6-89fd-12529bf3b8cf","year":2017},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.341319Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:b6c7afa88f9e213508263154c99273b4700d323990b5d0868f41a293c62b8845","observation_id":"1cba8519-29cb-44e7-beb5-2e893ce4d370","resolution":{"observed_at":"2026-08-09T05:02:13.613347Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.599866Z","title":"We also compare the behavior of agents learned from PVP and TD3 baseline","venue":null,"work_id":"72283d4d-5172-4d41-91ca-2d823b1e2b2f","year":null},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.345404Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:ffb03eb3a8b4705195d91524aa196f3203e9ef382ab798653b5c394ccc8a5d16","observation_id":"e950529c-f767-45f5-9422-d19e5ac28403","resolution":{"observed_at":"2026-08-09T05:02:13.603389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.588647Z","title":"We present the behavior comparison between PVP and TD3 baseline","venue":null,"work_id":"a0e88554-f43c-4522-a693-0e7a6e88eb6d","year":null},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.349364Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:419a6642117b29dbd0b7a3c96156353e3a555dc00db2870e316d1d617954a11c","observation_id":"6068c2d7-ed81-48b2-89b1-a6aefef1393f","resolution":{"observed_at":"2026-08-09T05:02:13.592962Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.577546Z","title":"PVP performs well in GTA V and can drive smoothly on the highway","venue":null,"work_id":"ef0f3360-605d-4b85-aad2-299bd279b978","year":null},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.353351Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:9e0ad117eedf84cc45c538df45933edde25e135c69da3426de1428789c6845fb","observation_id":"9bf4e8f1-30c6-4542-af09-7e41c28670bf","resolution":{"observed_at":"2026-08-09T05:02:13.581248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.567098Z","title":"If the agent drives in the wrong way then the displace- ment reward will be multiplied by −1","venue":null,"work_id":"0f67d18e-29fe-4dd4-841e-e3d7e6052135","year":null},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.357519Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:aa85bbe9d5e7d2fa4ab6bb864520150f600b4e80e652e455abae617ac0b6e8d2","observation_id":"c5bcf62e-6cee-4520-b026-ec406bf02213","resolution":{"observed_at":"2026-08-09T05:02:13.570786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.556232Z","title":"If the agent drives in wrong way then the speed reward will be multiplied by −1","venue":null,"work_id":"fab9506f-9d65-40b5-adeb-c300af6a91de","year":null},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.361332Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:354be659ee9afa04b60264b22c865071868c4148b5cab836e8442dc3c7f94301","observation_id":"c52ab57e-f03e-4c80-a06c-8884e315b45d","resolution":{"observed_at":"2026-08-09T05:02:13.559970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.545490Z","title":"Otherwise, it is 0","venue":null,"work_id":"41c59bce-9c44-4f4c-a115-041906186045","year":null},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.365174Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:798de1d9912f3c511625359c172df7e78faa07eec59402563b3b8ae96c17b698","observation_id":"17d20be8-c420-46e8-acdc-458c05267c24","resolution":{"observed_at":"2026-08-09T05:02:13.549236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T05:02:13.531639Z","title":"At that step, we set Rdisp = Rspeed = Rcollision = 0and assign Rterm according to the terminal state","venue":null,"work_id":"a352174a-ef36-49c2-ba23-d863a018f606","year":null},"citing_paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-09T05:02:13.369371Z"},"links":{"citing_paper":"/paper/2502.03369"},"observation_digest":"sha256:d53947a06e81c58d430595ad3ceca117d0eb99b69f58b4eb0f194320333626c1","observation_id":"f0fc79af-f7b3-490a-81a4-4890185e7071","resolution":{"observed_at":"2026-08-09T05:02:13.537874Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.03369","last_updated":"2025-02-05T17:07:37Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T12:36:37.004381Z","submitted_at":"2025-02-05T17:07:37Z","title":"Learning from Active Human Involvement through Proxy Value Propagation"},"reference_resolution":{"displayed":63,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":46},"total_outbound_references":63},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 63 of 63 outbound references and 0 inbound Pith citation observations for arXiv:2502.03369."}