{"as_of":"2026-08-04T11:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5b6c3294d52db1b49543be8f5eaa8dc54756853fb53796d5d82a0ed6d620c9a5","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-17T13:17:42.848578Z","state":"measured"},{"denominator":89,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":89,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":29,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T08:15:59.314080Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T06:15:00.866473Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2403.17297","last_updated":"2024-03-26T00:53:24Z","snapshot_observed_at":"2026-08-02T11:10:24.263044Z","submitted_at":"2024-03-26T00:53:24Z","title":"InternLM2 Technical Report","version":1},"reference_index":156,"source":"arxiv_source","source_observed_at":"2026-05-15T11:44:38.066501Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2403.17297"},"observation_digest":"sha256:6434cd84bb90bff145b709a615b0b405d6bddc9a98ec56af8e0e4a8086790feb","observation_id":"a5fec650-f8f9-465a-b4eb-ff755aa6ef86","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2411.10442","last_updated":"2025-04-07T09:09:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-15T18:59:27Z","title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","version":2},"reference_index":118,"source":"pdf_text","source_observed_at":"2026-05-16T09:16:17.150383Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2411.10442"},"observation_digest":"sha256:862ed35cba76f299bb45d66a86661abcaf907cb8d5631d2a162a93f3615cb0c3","observation_id":"c7d83599-890a-4bf7-885e-0fd052e2e2da","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2503.12575","last_updated":"2026-04-05T16:16:05Z","snapshot_observed_at":"2026-07-06T20:53:31.919637Z","submitted_at":"2025-03-16T17:06:00Z","title":"BalancedDPO: Adaptive Multi-Metric Alignment","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-22T23:37:55.154902Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2503.12575"},"observation_digest":"sha256:f6bee4239eb94416866678520d18e5bef349ebdf48d060d27fdc043efe27ac7c","observation_id":"d8732b1e-acf2-448e-b000-e2ce46e13a6d","resolution":{"observed_at":"2026-05-22T23:42:16.511659Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"reference_index":185,"source":"pdf_text","source_observed_at":"2026-05-10T11:58:58.660564Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2508.18265"},"observation_digest":"sha256:fe1facd60a09a10e4226f0c074bfa65f9677fe62f720e5b02f0de47e4e6434b6","observation_id":"aecf258c-2f26-4c3b-9268-43ef437e0746","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-08-04T08:15:59.314080Z","title":"Secrets of rlhf in large language models part i: Ppo","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.21978","last_updated":"2026-06-18T16:29:38Z","snapshot_observed_at":"2026-08-04T08:15:30.884838Z","submitted_at":"2025-10-24T19:08:48Z","title":"Beyond Reasoning Gains: Mitigating General-Capability Forgetting in Large Reasoning Models","version":2},"reference_index":115,"source":"arxiv_source","source_observed_at":"2026-08-04T08:15:59.314080Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2510.21978"},"observation_digest":"sha256:a923c08f762dce90015aacdb0a2e2bd2d9bdeffb6f2e55224b4c555c4d8de9a9","observation_id":"079b57ab-66c1-488c-aa25-c81a6c23f8fc","resolution":{"observed_at":"2026-08-04T08:15:59.314080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-08-04T07:21:37.066762Z","title":"Secrets of RLHF in large language models part I: PPO","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.26707","last_updated":"2026-07-15T04:49:02Z","snapshot_observed_at":"2026-08-04T07:10:35.135774Z","submitted_at":"2025-10-30T17:09:09Z","title":"Value Drifts: Tracing Value Alignment During LLM Post-Training","version":2},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-04T07:21:37.066762Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2510.26707"},"observation_digest":"sha256:2ddc9bd6ecdd6cc1a126f9994e60a80479dfa5a3b6ef5855442da88c4d583741","observation_id":"79a032b9-b362-4938-b476-819ec8889aea","resolution":{"observed_at":"2026-08-04T07:21:37.066762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2601.04068","last_updated":"2026-05-20T06:08:56Z","snapshot_observed_at":"2026-07-06T22:41:01.254450Z","submitted_at":"2026-01-07T16:32:17Z","title":"Mind the Generative Details: Direct Localized Detail Preference Optimization for Video Diffusion Models","version":3},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-16T16:25:03.743594Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2601.04068"},"observation_digest":"sha256:0b307e357b0066f4be1a3cc20918fe63797bf9118c7790200ef1f1186f9f815a","observation_id":"42533882-2aa0-4c01-8b88-00bab6aa5854","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2601.04068","last_updated":"2026-05-20T06:08:56Z","snapshot_observed_at":"2026-07-06T22:41:01.254450Z","submitted_at":"2026-01-07T16:32:17Z","title":"Mind the Generative Details: Direct Localized Detail Preference Optimization for Video Diffusion Models","version":4},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-21T16:01:52.150950Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2601.04068"},"observation_digest":"sha256:f001d1e0fe4df91a5bd659342dc0e4e02eda7e386e0c09a5e64137b37cd05df5","observation_id":"a37ad172-a3aa-42b7-9bae-321b09a92d3d","resolution":{"observed_at":"2026-05-21T16:04:14.715694Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2603.00774","last_updated":"2026-04-05T12:37:19Z","snapshot_observed_at":"2026-08-01T21:35:47.520638Z","submitted_at":"2026-02-28T18:39:19Z","title":"Structure Matters: Evaluating Multi-Agents Orchestration in Generative Therapeutic Chatbots","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-15T17:59:40.966701Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2603.00774"},"observation_digest":"sha256:e537dc1f1c4083562470bffb6fd758e4c6d1d5f75022637a6f98c8483a98111f","observation_id":"16af6309-c547-4d77-8f62-08b5bcb033f5","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2603.12631","last_updated":"2026-04-27T03:20:35Z","snapshot_observed_at":"2026-08-01T19:17:30.395711Z","submitted_at":"2026-03-13T04:04:17Z","title":"Joint Optimization of Multi-agent Memory System","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-15T12:12:35.056095Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2603.12631"},"observation_digest":"sha256:af55a3b6b29ea27f02f49a737e737275b7303d0dae781501a5ca8c47fe635df9","observation_id":"6744e1cf-4792-4f3f-9a6b-98c5e803d947","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2604.07941","last_updated":"2026-04-16T04:43:04Z","snapshot_observed_at":"2026-08-02T07:24:04.075021Z","submitted_at":"2026-04-09T08:00:37Z","title":"Large Language Model Post-Training: A Unified View of Off-Policy and On-Policy Learning","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-05-10T18:28:58.515666Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2604.07941"},"observation_digest":"sha256:46b36aabcfde71619eb869c93091f866bb905603aed32efe9a327d64044945ef","observation_id":"fb0571e0-0a13-48c9-986c-0ce4a464119f","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2604.11446","last_updated":"2026-04-13T13:28:12Z","snapshot_observed_at":"2026-07-06T22:59:49.614780Z","submitted_at":"2026-04-13T13:28:12Z","title":"Low-rank Optimization Trajectories Modeling for LLM RLVR Acceleration","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T16:07:20.180349Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2604.11446"},"observation_digest":"sha256:978edacf8f2cab84125478d6219d4b2bef98b12fe6cd2c6d5169bc03f45153f2","observation_id":"0fb9b332-5e56-44d2-8074-21af8a30aee0","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2604.17396","last_updated":"2026-04-19T11:59:58Z","snapshot_observed_at":"2026-08-01T18:23:32.699442Z","submitted_at":"2026-04-19T11:59:58Z","title":"Representation-Guided Parameter-Efficient LLM Unlearning","version":1},"reference_index":203,"source":"arxiv_source","source_observed_at":"2026-05-10T06:01:46.885030Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2604.17396"},"observation_digest":"sha256:2b7e8f2d5b47bc6130f96792eb44e4b4aa1fa08c9d571c6d062e9d4c65a67a8d","observation_id":"7a6355db-9581-421a-af63-35552903dee1","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2604.19485","last_updated":"2026-04-21T14:07:39Z","snapshot_observed_at":"2026-07-06T23:06:07.161558Z","submitted_at":"2026-04-21T14:07:39Z","title":"EVPO: Explained Variance Policy Optimization for Adaptive Critic Utilization in LLM Post-Training","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-10T03:00:10.404757Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2604.19485"},"observation_digest":"sha256:0af78f91ec0e231d99e3195a6a167d0d066bc07eef399d03b2011a3ae1d67d2d","observation_id":"337a1bbe-7e7d-4c1c-99fd-a31b223f6a84","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2604.28020","last_updated":"2026-05-29T15:07:05Z","snapshot_observed_at":"2026-07-06T23:13:24.960159Z","submitted_at":"2026-04-30T15:39:09Z","title":"Cost-Aware Learning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-07T05:11:01.131590Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2604.28020"},"observation_digest":"sha256:5ac2ab6d9667e1f64948af0cd25c473a2d99596e1194c8904ce462b4ec06bd56","observation_id":"dd53dbfc-81d9-4f7e-a992-c3e160139285","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2605.00347","last_updated":"2026-05-01T02:05:56Z","snapshot_observed_at":"2026-07-06T23:13:47.082550Z","submitted_at":"2026-05-01T02:05:56Z","title":"Odysseus: Scaling VLMs to 100+ Turn Decision-Making in Games via Reinforcement Learning","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-05-09T20:22:58.061772Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2605.00347"},"observation_digest":"sha256:f9115e8a9e42dbcbf404411ac945482a92f232a36f0bbac2da7c41f03a9ae105","observation_id":"8ddb67de-8805-48a4-a6b6-7871e39b5c86","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2605.07522","last_updated":"2026-05-08T09:53:33Z","snapshot_observed_at":"2026-08-01T17:24:03.411313Z","submitted_at":"2026-05-08T09:53:33Z","title":"WeatherSyn: An Instruction Tuning MLLM For Weather Forecasting Report Generation","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-11T01:55:07.218958Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2605.07522"},"observation_digest":"sha256:30097cb35773b6fdc33a140cfd6a18ee86b0c0b6a782dc45a905a6636d9f99d9","observation_id":"bc8c70f2-f20d-4ef4-a46b-8675df69faad","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2605.08378","last_updated":"2026-05-08T18:36:25Z","snapshot_observed_at":"2026-08-02T18:04:56.183678Z","submitted_at":"2026-05-08T18:36:25Z","title":"Reinforcement Learning for Scalable and Trustworthy Intelligent Systems","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-12T01:47:40.772146Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2605.08378"},"observation_digest":"sha256:67db0d89e3389d4be431e33630746deec02ade57af6ed2d5b3d33148ce8d5217","observation_id":"140bbb1e-2e88-4f88-9cd4-638c79a8512e","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2605.08665","last_updated":"2026-06-03T08:11:51Z","snapshot_observed_at":"2026-07-06T23:20:52.521341Z","submitted_at":"2026-05-09T04:07:16Z","title":"Hint Tuning: Less Data Makes Better Reasoners","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-12T00:54:13.146373Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2605.08665"},"observation_digest":"sha256:23d7374fccf5fa299978b8582c35aac3a0460ec1539de0443187c509060137e7","observation_id":"637d5dbf-03c7-4652-a35d-d3da5fb642ba","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2605.08665","last_updated":"2026-06-03T08:11:51Z","snapshot_observed_at":"2026-07-06T23:20:52.521341Z","submitted_at":"2026-05-09T04:07:16Z","title":"Hint Tuning: Less Data Makes Better Reasoners","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-30T23:34:43.785312Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2605.08665"},"observation_digest":"sha256:5a5ccd36c3fc596e1f6041036c887aa907c6a02e35166c01c5bdcdab7785e3f2","observation_id":"b199c1fa-6a92-43f7-a2c0-b59a76b8d585","resolution":{"observed_at":"2026-06-30T23:35:07.069222Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2605.11235","last_updated":"2026-05-11T20:50:29Z","snapshot_observed_at":"2026-07-06T23:23:06.755816Z","submitted_at":"2026-05-11T20:50:29Z","title":"Internalizing Curriculum Judgment for LLM Reinforcement Fine-Tuning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T02:19:27.345348Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2605.11235"},"observation_digest":"sha256:0739ef27c96816b364d92d05748f4e3ca834ce11ec67ec052eb72958ebb867ea","observation_id":"671a9382-df45-42b8-b36b-0bc2c338e1a5","resolution":{"observed_at":"2026-05-17T13:17:43.137115Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2605.27788","last_updated":"2026-05-27T00:11:31Z","snapshot_observed_at":"2026-08-02T05:35:24.628664Z","submitted_at":"2026-05-27T00:11:31Z","title":"Knowing When to Ask: Segment-Level Credit Assignment for LLM Tool Use","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-29T13:49:58.677299Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2605.27788"},"observation_digest":"sha256:d7729114a2d2aea07ec773196b84c60b9936376df98e45d58619e67cab1430d7","observation_id":"fa0af2ee-e070-4316-9378-9308c1a43bc9","resolution":{"observed_at":"2026-06-29T13:53:28.605328Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2606.03070","last_updated":"2026-06-23T05:24:51Z","snapshot_observed_at":"2026-08-03T22:34:21.108001Z","submitted_at":"2026-06-02T03:00:34Z","title":"ASymPO: Asymmetric-Scale Policy Optimization for Asynchronous LLM Post-Training Without Behavior Information","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-06-28T11:11:41.433882Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2606.03070"},"observation_digest":"sha256:7dcc07df172c6f2a267c87b18875a74753cc41cd2d5e5bea3e5dc07e93fae33a","observation_id":"c6cc844f-c89b-4570-8f6c-5567af0040d1","resolution":{"observed_at":"2026-07-02T02:06:27.471531Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2606.09043","last_updated":"2026-06-08T05:24:15Z","snapshot_observed_at":"2026-08-02T05:26:25.857871Z","submitted_at":"2026-06-08T05:24:15Z","title":"DynaCF: Mitigating Shortcut Learning in Reward Models via Dynamic Counterfactual Sensitivity","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-06-27T17:26:17.072017Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2606.09043"},"observation_digest":"sha256:1199e3c74a665a9d38b62a15e0f0fbca6147a783acfbdacdf5bcb89a9078d778","observation_id":"c96ae485-9b02-4d5f-9a56-6fbc903912cd","resolution":{"observed_at":"2026-06-27T17:31:06.955538Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2606.21943","last_updated":"2026-06-20T08:20:41Z","snapshot_observed_at":"2026-07-06T23:56:54.959593Z","submitted_at":"2026-06-20T08:20:41Z","title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","version":1},"reference_index":274,"source":"pdf_text","source_observed_at":"2026-06-26T12:15:08.304150Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2606.21943"},"observation_digest":"sha256:73b749358dafb9d4e51d48479f1f2b4c34a195b40b7261e282f0e542e0161d75","observation_id":"802fb54f-9835-46d3-9103-0cf7a54865d2","resolution":{"observed_at":"2026-07-04T08:09:40.639382Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":"2307.04964","doi":"10.48550/arxiv.2307.04964","metadata_source":"pith","pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","venue":"cs.CL","work_id":"e3d402e3-52dc-466f-998e-a8932c1b0bfc","year":2023},"citing_paper":{"arxiv_id":"2606.23950","last_updated":"2026-06-22T21:22:42Z","snapshot_observed_at":"2026-08-03T10:32:13.705652Z","submitted_at":"2026-06-22T21:22:42Z","title":"DivRL: Disentangled Self-Similarity Rewards for Diverse Subject-Driven Generation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-26T08:37:11.052296Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2606.23950"},"observation_digest":"sha256:b72a65eb06636393d426f6885e92944ed042ea1a74911dbb9aa970189b7be840","observation_id":"87039985-0caf-465a-a8fa-279fcfc84daf","resolution":{"observed_at":"2026-07-04T10:39:45.593093Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-14T15:37:01.391649Z","title":"Secrets of rlhf in large language models part i: Ppo","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.09796","last_updated":"2026-07-20T03:41:16Z","snapshot_observed_at":"2026-08-02T08:00:28.789067Z","submitted_at":"2026-07-09T09:20:25Z","title":"Metadata-Free Meta-Reweighted Direct Preference Optimization under Noisy Preference Labels","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-14T15:37:01.391649Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2607.09796"},"observation_digest":"sha256:90cc001477d50367eccef4c890ed713629a3973e83c3e14297ab1f2cd0f4cc52","observation_id":"e97938e4-c485-40ed-901f-838cdbb0a09c","resolution":{"observed_at":"2026-07-14T15:37:01.391649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-08-02T08:00:53.466595Z","title":"Secrets of rlhf in large language models part i: Ppo,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.09796","last_updated":"2026-07-20T03:41:16Z","snapshot_observed_at":"2026-08-02T08:00:28.789067Z","submitted_at":"2026-07-09T09:20:25Z","title":"Metadata-Free Meta-Reweighted Direct Preference Optimization under Noisy Preference Labels","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-02T08:00:53.466595Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2607.09796"},"observation_digest":"sha256:9febd5ce9b77fb956882c2ebfecec95d00ec246685ae5afa345659344bdcf2b5","observation_id":"6feb314c-d597-4fd9-8763-ced9bfa9ad1f","resolution":{"observed_at":"2026-08-02T08:00:53.466595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.04964","snapshot_observed_at":"2026-07-31T01:31:10.167491Z","title":"arXiv preprint arXiv:2307.04964 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27610","last_updated":"2026-07-30T02:55:02Z","snapshot_observed_at":"2026-08-02T23:35:59.061631Z","submitted_at":"2026-07-30T02:55:02Z","title":"Kalman Meets Curriculum: Efficient Dynamic Prompt Selection for Adaptive RL Finetuning","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-07-31T01:31:10.167491Z"},"links":{"cited_paper":"/paper/2307.04964","citing_paper":"/paper/2607.27610"},"observation_digest":"sha256:8142450aca0d0537213c3e3dd92d3288256008fa9dad7e84ac560443ff769d95","observation_id":"5b45bfb3-0e3c-4249-b41b-dc6be31f009c","resolution":{"observed_at":"2026-07-31T01:31:10.167491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2307.04964/citation-record","integrity":"/paper/2307.04964/integrity","json":"/paper/2307.04964/citation-record.json","paper":"/paper/2307.04964"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-07-11T03:47:48.508181Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:6f557d0e173e90c72f76df85bcb7421a4c38e5ef39ed4a5e2b70eb5b3a24ef6c","observation_id":"8c75b798-b3e1-4489-ab0a-e23faef0efde","resolution":{"observed_at":"2026-05-17T13:17:42.968020Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"17fc14bb-5dff-4a72-af61-425c91b07479","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:275140b97aa3185da1a54644a2b0cdcd303ff59f8bb40d849bca18f6d60c9848","observation_id":"c41d7993-64ca-4374-af59-1d3d6872d061","resolution":{"observed_at":"2026-05-17T13:17:43.093995Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T07:26:54.787661Z","title":"Gpt-4 technical report","venue":null,"work_id":"388f534c-855a-4366-b933-f07bf3e2db5f","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:335c4477c0ff2cf14c695eb86118497a9bab4634efd1b65f5d90c8cd8c86244b","observation_id":"69dfe1fb-bb39-4474-a9a0-4fb6b7b47135","resolution":{"observed_at":"2026-05-17T13:17:43.096993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-04T11:07:48.536954Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":"2303.18223","doi":"10.18653/v1/d16-1080","metadata_source":"pith","pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"A Survey of Large Language Models","venue":"cs.CL","work_id":"de1b42b5-4a0a-4b1f-8c78-1f7fe21be6c9","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:9196a46e1038a87d6b2f17683aedce7ea86dc64e73ee19c0f3dd3fce8ea5f16a","observation_id":"b3233dc6-fc14-453f-819c-bc31f858b810","resolution":{"observed_at":"2026-05-17T13:17:42.893700Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9010afde-4504-4219-a609-1afdc37b81a3","year":1901},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:00f464eb7c2c32e561b6e03a113c409f2747271f284584cda180610b9342fa93","observation_id":"e5c94e01-3d74-488e-8ef2-b21f6025527d","resolution":{"observed_at":"2026-05-17T13:17:43.100456Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03277","last_updated":"2023-04-06T17:58:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-06T17:58:09Z","title":"Instruction Tuning with GPT-4","version":1},"cited_work":{"arxiv_id":"2304.03277","doi":"10.48550/arxiv.2304.03277","metadata_source":"pith","pith_arxiv_id":"2304.03277","snapshot_observed_at":"2026-07-10T20:37:34.448563Z","title":"Instruction Tuning with GPT-4","venue":"cs.CL","work_id":"fd515477-f9f1-48aa-9feb-a3308e7656bb","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/2304.03277","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:35d00a5e89f741329612e5c0c6a17e4684a7eda478dd87d4500bed6cd810ac49","observation_id":"9ef53359-f728-4091-936a-4cd9ea33fbb8","resolution":{"observed_at":"2026-05-17T13:17:42.956775Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gulrajani, T","venue":null,"work_id":"3c97fdc6-7287-4eca-b950-04d124014b50","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:1f99e92715bc6302c239600e9c515900db5502ade9405f6ee666eaeb6ad8e650","observation_id":"9420be41-fa2d-4017-8d71-c27338517d79","resolution":{"observed_at":"2026-05-17T13:17:43.107279Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11903","last_updated":"2023-01-10T23:07:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-28T02:33:07Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":"2201.11903","doi":"10.48550/arxiv.2201.11903","metadata_source":"pith","pith_arxiv_id":"2201.11903","snapshot_observed_at":"2026-07-10T21:57:37.821767Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","venue":"cs.CL","work_id":"d1cf6693-a082-403c-ada9-dac7b96341f9","year":2022},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/2201.11903","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:63c1c712b60bf19677ecfb0afc263d4b60207e4e01265ee4f6b0366c7ae7532f","observation_id":"450942f6-a1f2-484d-86d8-91000a1df7a1","resolution":{"observed_at":"2026-05-17T13:17:42.913043Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:13.648188+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:13.648188+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03378","last_updated":"2023-03-06T18:58:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-06T18:58:06Z","title":"PaLM-E: An Embodied Multimodal Language Model","version":1},"cited_work":{"arxiv_id":"2303.03378","doi":"10.48550/arxiv.2303.03378","metadata_source":"pith","pith_arxiv_id":"2303.03378","snapshot_observed_at":"2026-07-11T00:27:51.904332Z","title":"PaLM-E: An Embodied Multimodal Language Model","venue":"cs.LG","work_id":"5b99811a-1d93-47e2-9d59-f4045a0b74a2","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/2303.03378","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:be1d4dd92166e3af834346a359c878b531e430ddd350f46100798da12acc00f4","observation_id":"67047b39-235c-4e3a-b88e-505211a4a1c2","resolution":{"observed_at":"2026-05-17T13:17:42.937746Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-09T19:19:06.044727+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T19:19:06.044727+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03442","last_updated":"2023-08-06T00:21:19Z","snapshot_observed_at":"2026-07-06T15:13:13.695535Z","submitted_at":"2023-04-07T01:55:19Z","title":"Generative Agents: Interactive Simulacra of Human Behavior","version":2},"cited_work":{"arxiv_id":"2304.03442","doi":"10.1001/jamapsychiatry.2022.0609","metadata_source":"pith","pith_arxiv_id":"2304.03442","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Generative Agents: Interactive Simulacra of Human Behavior","venue":"cs.HC","work_id":"01f7ddaa-284a-441a-be87-921aad4dc54b","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/2304.03442","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:8b4dc25661879a2bf4b5be4856314143673a3f934f95ebef42f50c6b7ed2f583","observation_id":"063d1ef0-7e09-4872-a931-8d427161e3da","resolution":{"observed_at":"2026-05-17T13:17:42.947701Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-24T05:54:33.103795+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T05:54:33.103795+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"bbffa7ed-b95d-4325-99b3-05de8d03483c","year":2021},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:469188ec225adedb5185b6e477f0ce071ccfcd0adc34c0f8723fdae12131277d","observation_id":"90db619f-d530-4d5e-8a98-e01cadefdf5a","resolution":{"observed_at":"2026-05-17T13:17:43.110813Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.08239","last_updated":"2022-02-10T16:30:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-20T15:44:37Z","title":"LaMDA: Language Models for Dialog Applications","version":3},"cited_work":{"arxiv_id":"2201.08239","doi":"10.48550/arxiv.2201.08239","metadata_source":"pith","pith_arxiv_id":"2201.08239","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"LaMDA: Language Models for Dialog Applications","venue":"cs.CL","work_id":"1b66d0a5-f6ae-4332-8025-c662dc64b238","year":2022},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/2201.08239","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:9124508d7a1487d3751e1f93e2245670267414df42c92f1d3141a1b3ea355559","observation_id":"8bbb306d-edac-4934-99a0-9bf0c388b194","resolution":{"observed_at":"2026-05-17T13:17:42.887801Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"74c5d688-65b9-4c7c-9488-aded4bf982e5","year":2021},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:17ec02e9c57792f6309d58f070b2517edda6e16e514164dc5937a29c1d352283","observation_id":"362e272c-ad1b-43b8-a6d9-1cff94be1aec","resolution":{"observed_at":"2026-05-17T13:17:43.113927Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":"2108.07258","doi":"10.1016/j.specom.2008.12.003","metadata_source":"pith","pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"On the Opportunities and Risks of Foundation Models","venue":"cs.LG","work_id":"a18039e9-928d-47c9-a836-32656a71bf71","year":2021},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:bcde89ba1f361c14da8d44246a0684cdd5a63dcba43c1a9563a9395f11b14e50","observation_id":"959517ac-717e-466e-92a3-f0789164ed41","resolution":{"observed_at":"2026-05-17T13:17:42.899373Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:22:59.787075+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:22:59.787075+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Planning for agi and beyond","venue":null,"work_id":"b7ae0fac-e707-4ee1-8222-f429ad46d6a7","year":2022},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:3f3a21387da12855b5c718d2ffd84a4c43bb7177c14d97df093ec98603c2e708","observation_id":"1be1a84e-7f77-4bb1-920b-74cd8fd6b9ae","resolution":{"observed_at":"2026-05-17T13:17:43.117595Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":"2203.02155","doi":"10.1007/s00354-022-00198-8","metadata_source":"pith","pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Training language models to follow instructions with human feedback","venue":"cs.CL","work_id":"52aff42f-4fa9-4fcf-bdb3-1459b9bebf65","year":2022},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:35ef562313493ac61f785c8bded8d79a40fe13032ce49f7521ba57d10e954598","observation_id":"3e0e8d1b-3579-4a52-87f2-c3192cdb82db","resolution":{"observed_at":"2026-05-17T13:17:42.920431Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":"2204.05862","doi":"10.1016/j.respol.2005.01.014","metadata_source":"pith","pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","venue":"cs.CL","work_id":"a1f2574b-a899-4713-be60-c87ba332656c","year":2022},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:c61d9b3754bf7cac700cbd22ef56823b8dc2c482240cfd19de67d59884508d42","observation_id":"88012d84-56eb-4107-816c-7c24261f9c3d","resolution":{"observed_at":"2026-05-17T13:17:42.931062Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Open-Chinese-LLaMA: Chinese large language model base generated through incremental pre-training on chinese datasets","venue":null,"work_id":"63a86e62-c3c9-4fec-88f8-102d44d3da35","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:31871306704d5b734673c5c186ac53a569ab88ed6e2084f52907c70a69533889","observation_id":"e90edba7-e98e-4085-9bd4-d107eedad9a1","resolution":{"observed_at":"2026-05-17T13:17:43.121043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"545a432a-0916-44bb-8116-630b8d688eab","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:1878923ed7468a320c63993b345f52b9ffb5159c1494ce637db203f36dd3809f","observation_id":"a52d831f-0e06-46f9-8460-32f56fa5f1b9","resolution":{"observed_at":"2026-05-17T13:17:43.123770Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f1b08f70-183a-45b9-851c-ba86c28c2d6f","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:691aa4e1be58fe398536305cd5d2401bcdb1d47f84c055e8a1265a223f5b4b0b","observation_id":"27c5086e-590a-45ed-b29b-b7a2c2194e99","resolution":{"observed_at":"2026-05-17T13:17:43.126563Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Belkada, K","venue":null,"work_id":"b469afb7-d418-47b2-aa9c-96e88f2730c0","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:05a4c5971c4785fa7f0c9b7e2a0fb41b466c2579102306d9e2d06dea6db8eee7","observation_id":"abcd338c-bf78-4d39-aa02-6e24a222c534","resolution":{"observed_at":"2026-05-17T13:17:43.130019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a275afdf-0c73-4031-ab22-e78009c81c68","year":2017},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:790e04e33dc5de01b6c5a5dc0dff3022529cf5eb36f64508fe800b25756273f0","observation_id":"b8987e0e-67b4-4a9a-9d4a-8b52f2912a79","resolution":{"observed_at":"2026-05-17T13:17:43.133088Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a0187bc4-160f-4ffc-bb95-cecf6bf98488","year":2017},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:49f7eb6a024b7d30a703b7e99eba517034773c012a22708b602369e41e2f9d58","observation_id":"a75b9212-8747-4e18-bc3d-0b6d587a30fe","resolution":{"observed_at":"2026-05-17T13:17:43.135888Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08593","last_updated":"2020-01-08T23:02:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-09-18T17:33:39Z","title":"Fine-Tuning Language Models from Human Preferences","version":2},"cited_work":{"arxiv_id":"1909.08593","doi":"10.48550/arxiv.1909.08593","metadata_source":"pith","pith_arxiv_id":"1909.08593","snapshot_observed_at":"2026-07-10T19:37:34.037876Z","title":"Fine-Tuning Language Models from Human Preferences","venue":"cs.CL","work_id":"4f54aad1-f3b6-404f-b9c7-e21ba0a33b99","year":2019},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/1909.08593","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:79c91a28fab4976679100288cd9163e6f732b3780dc4f169ed08792372e8b9dc","observation_id":"9ffa8ae5-82d4-4c32-a140-ccf029b722e0","resolution":{"observed_at":"2026-05-17T13:17:42.904891Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-18T08:21:06.631932+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-18T08:21:06.631932+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ouyang, J","venue":null,"work_id":"360f00fe-9783-488a-8709-6321417a9dc2","year":2020},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:795d6485c089160c4c95b64dd48bca49934dae56be3668ceb031d3f01fbd1194","observation_id":"c4918f68-6023-4133-9914-7d79d4998a27","resolution":{"observed_at":"2026-05-17T13:17:42.970881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Kadavath, S","venue":null,"work_id":"3a2f6263-6f4b-4905-8bb3-5d31eb9fd79d","year":2022},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:96c680fd49ceed826a6d5a0f9eb614f1b6936c6fb9b0c497dd60b8719efdee64","observation_id":"13b42ce3-24b3-4375-ad50-ad35d9adde35","resolution":{"observed_at":"2026-05-17T13:17:42.973536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.00861","last_updated":"2021-12-09T21:40:22Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-12-01T22:24:34Z","title":"A General Language Assistant as a Laboratory for Alignment","version":3},"cited_work":{"arxiv_id":"2112.00861","doi":"10.48550/arxiv.2112.00861","metadata_source":"pith","pith_arxiv_id":"2112.00861","snapshot_observed_at":"2026-07-10T15:37:20.432112Z","title":"A General Language Assistant as a Laboratory for Alignment","venue":"cs.CL","work_id":"a43f9ea0-01be-47d5-b8ee-a1a9f73381c5","year":2021},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/2112.00861","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:cceea2f2088c7aad40a9abdb032575cde784e4fc48bbb853d7dd6ea2614cb9bd","observation_id":"e420e98f-deb5-4b75-b940-f55afb5930e4","resolution":{"observed_at":"2026-05-17T13:17:42.926079Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-20T18:52:13.802987+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T18:52:13.802987+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Raichuk, P","venue":null,"work_id":"3dafb585-eee3-4adc-a7b2-af779c2ae285","year":2021},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:231ad92ffd17680c03c8138a1fb89aa1f93ea53499b9b3b70422d0c08c4a508d","observation_id":"4b09fc6d-a2d2-40ba-8eaf-036cd0064c13","resolution":{"observed_at":"2026-05-17T13:17:42.976099Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ilyas, S","venue":null,"work_id":"94457686-36d7-4125-92f2-e8a5893f2e12","year":2020},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:c59d6bdc75acda39b8a1bb13327addce3fddbae746c0823a63060841748df653","observation_id":"bef6cfba-6e93-4065-b7e3-5e533fe3d3dd","resolution":{"observed_at":"2026-05-17T13:17:42.978598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.09751","last_updated":"2020-02-14T21:56:30Z","snapshot_observed_at":"2026-07-06T07:47:32.745963Z","submitted_at":"2019-04-22T07:17:18Z","title":"The Curious Case of Neural Text Degeneration","version":2},"cited_work":{"arxiv_id":"1904.09751","doi":"10.48550/arxiv.1904.09751","metadata_source":"pith","pith_arxiv_id":"1904.09751","snapshot_observed_at":"2026-07-11T02:27:47.673937Z","title":"The Curious Case of Neural Text Degeneration","venue":"cs.CL","work_id":"1ef2ec4a-db7b-42b2-9416-0f1bb628c3c8","year":2019},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/1904.09751","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:16462a7cfcca68c2e9ae4ae1a2d4e70d4548b228e57b120460952c83a0283dad","observation_id":"44eca8ee-1ff5-429a-aca5-9022f487e621","resolution":{"observed_at":"2026-05-17T13:17:42.942850Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-19T15:54:15.356125+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T15:54:15.356125+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1119786e-20a2-4519-b422-1f9d17b202c3","year":2016},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:8d97ae77b51764b2a0bd160e9b5dd24d5f3ae818b7b5c65881c49bd6f3463b7b","observation_id":"0451a93a-93d7-4eba-853e-49fe3e44faa5","resolution":{"observed_at":"2026-05-17T13:17:42.981207Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.00456","last_updated":"2019-07-08T17:21:46Z","snapshot_observed_at":"2026-07-06T08:03:53.351060Z","submitted_at":"2019-06-30T20:53:19Z","title":"Way Off-Policy Batch Deep Reinforcement Learning of Implicit Human Preferences in Dialog","version":2},"cited_work":{"arxiv_id":"1907.00456","doi":null,"metadata_source":"pith","pith_arxiv_id":"1907.00456","snapshot_observed_at":"2026-07-03T01:37:30.382383Z","title":"Way Off-Policy Batch Deep Reinforcement Learning of Implicit Human Preferences in Dialog","venue":"cs.LG","work_id":"42fcaa3e-0409-481b-9dd5-9a2c3be8a383","year":2019},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"cited_paper":"/paper/1907.00456","citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:ceb7174f524cf297534c769f441163b5385340e3534bcaab1345fe5e9661dbb6","observation_id":"b4ca9be8-c4da-4a10-8bf8-7f6f54702625","resolution":{"observed_at":"2026-05-17T13:17:42.952535Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Levine, P","venue":null,"work_id":"246db874-c153-4576-8ed6-fb06c32452e0","year":2015},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:47a83d5964ae3ce1f273b9d1ec43b3802e33fabb895292604f27d6116ea938e7","observation_id":"dbb6f39c-977c-4500-a387-4a29d5386b26","resolution":{"observed_at":"2026-05-17T13:17:42.983742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wolski, P","venue":null,"work_id":"522551a3-24cc-4127-aec8-7047faff2266","year":2017},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:e5e832df501fc909f50a776e371d6e50e05fdc8fd463a7bcc6cf271c63076e6c","observation_id":"dc6f1e59-ff0d-4fdb-a2b1-bc929e6d469b","resolution":{"observed_at":"2026-05-17T13:17:42.986362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c2a2a215-781c-456f-b0b7-2e776d05531e","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:e65f4324dd1bae3c2a90230d21b21bb969a91eba499d7dfd819febc95c003ba4","observation_id":"87fc0409-5d9d-4f22-b811-80b23c0f40e8","resolution":{"observed_at":"2026-05-17T13:17:42.989057Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Kavukcuoglu, D","venue":null,"work_id":"2cd78f51-482e-4e31-a5a5-c0593210a97e","year":2015},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:c61af1b8f77d9d27513643d2863b562d6b329129011757c7320a5f851912a38d","observation_id":"e0023d86-d8e4-4e5a-a8c3-f31381eacd33","resolution":{"observed_at":"2026-05-17T13:17:42.991580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"J., Yiyuan Yang.Easy RL: Reinforcement Learning Tutorial","venue":null,"work_id":"49b52188-3a0a-40f1-92f1-c566b7ed79f1","year":2022},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:c14e567b424f4d61cec8293416489eef89614fb069c2eb8a01714b14013e068f","observation_id":"67435465-f740-45c3-a8b6-d6ae66cd97f0","resolution":{"observed_at":"2026-05-17T13:17:42.994445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"McCann, L","venue":null,"work_id":"8ad6f880-af41-4db8-9ad1-bcac49671a02","year":2019},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:cbfe102e2ac7570acb8400a26c6f88f469a8a8d8ef287017f7066769e1690e5f","observation_id":"6b7da4a7-22f3-4533-a316-20192625f5b9","resolution":{"observed_at":"2026-05-17T13:17:42.998877Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8b594f46-7ffe-4d76-96ad-07e4ebbef34b","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:80dfc74b0f7b8a5c342701e5cc6f9a060c9789f034b3059a6f2c4bdbb24724e7","observation_id":"d0250bf3-b28c-4df3-bbea-c19bfef28078","resolution":{"observed_at":"2026-05-17T13:17:43.001939Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chiang, Y","venue":null,"work_id":"51233af2-ed44-4ec1-8418-f57adfaf3532","year":2023},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:603815e2d946b49f4a7ddb2920c5300fe4a6d53fe348a7728ba5901c393ca3dd","observation_id":"4674a8af-5b4c-41dd-bd60-2f696b8427a4","resolution":{"observed_at":"2026-05-17T13:17:43.010143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The idea is that these organisms could have survived the journey through space and then established themselves on our planet","venue":null,"work_id":"5dbe6117-6fc7-48d8-9526-4fee973867f2","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:dd635e4fc3661cfd5e1328b7a6c6b59db52a25f5f2a3113a1d034a5babbe3f02","observation_id":"bef4b476-4310-4bb5-9d01-a8599bd2ddd6","resolution":{"observed_at":"2026-05-17T13:17:43.014840Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Over time, these compounds would have organized themselves into more complex molecules, eventually leading to the formation of the first living cells","venue":null,"work_id":"5ba94830-d65f-44c4-b858-d64ebe904074","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:be8d2c0a4982af8d41da8e9784c25c0c347d0ac50c72257849d03e6a976465f6","observation_id":"fa60282d-836e-4e01-bd3a-50052dc98524","resolution":{"observed_at":"2026-05-17T13:17:43.021831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"These organisms were able to thrive in an environment devoid of sunlight, using chemical energy instead","venue":null,"work_id":"8b29fa72-c76f-4e46-9f89-970d617f9c74","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:390ca7ba35549cc188767d41950a80c419382bed2447edf12520f44b24e84008","observation_id":"3534006e-65a9-4099-b250-869414f4e831","resolution":{"observed_at":"2026-05-17T13:17:43.025162Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7d83a136-d2ad-43af-b4e4-2c6e0b0a4461","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:8fe0d00cdc507d7d15b8b9f2e80ee3e04172fde7a14aa5d53dc9824a77d04791","observation_id":"6f741e3f-c0db-416a-b8d4-47ccb5660883","resolution":{"observed_at":"2026-05-17T13:17:43.038985Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"6607f7ab-a543-4427-a544-3d35d4c978c1","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:5b78c73b99761d43b8b82dba2872770a27eb34de28a3f8132b54caf97fa8dedd","observation_id":"4e4b74ec-10a9-4637-8852-ca86eb6baa70","resolution":{"observed_at":"2026-05-17T13:17:43.042349Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0918ea84-cc41-47d0-b834-05899743ab0b","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:3c0ab08ee7ad64b118a59793b3e2ec8c183fd689cfd8f37504a9aea6f32d115a","observation_id":"f82109a0-35fe-4672-a8a3-ff0b5a9e003f","resolution":{"observed_at":"2026-05-17T13:17:43.045564Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"28ba3438-e3ad-4825-b866-ea0fd896128a","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:28f350ed7d6af5937c3c3ad0e3c2743082336e0d78499d6ad77695c0e9d39ea7","observation_id":"3ec9ff71-fb82-4f7a-baa1-956c70c3d845","resolution":{"observed_at":"2026-05-17T13:17:43.048597Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4ac93ecb-ae35-4859-9a7c-a5c37ba88e9b","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:90f1340f03a5387d4b573e461ce4642f6c941b2d883ab2db2b1ff622338b2098","observation_id":"4a0ffa7c-30a9-44c4-97c2-2f913f998ce0","resolution":{"observed_at":"2026-05-17T13:17:43.051474Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This is just one example of many scams that prey on vulnerable older adults","venue":null,"work_id":"9fbe760b-22bf-4e1e-9aa3-740a2fde7fc1","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:76bc22a5a7324931fa06357ef946c9195fb5a39500e09fb5ab7ce44f99e04a27","observation_id":"035e814d-230c-49e8-a725-b013ab446ec8","resolution":{"observed_at":"2026-05-17T13:17:43.054819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4358ea61-78b5-4b30-8e52-0006d659e311","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:d886dc9bee5788d7df885e3038a02105edded9f0099a6b52a5cb203525b477fd","observation_id":"f3435e97-6136-4655-91ad-3be6fc2c52c3","resolution":{"observed_at":"2026-05-17T13:17:43.057937Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"It can result in legal trouble, damage to your reputation, and harm to your relationships with others","venue":null,"work_id":"cd11a5d9-8f04-4a32-a3dd-a60c7ffb32d2","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:cd710725864ee6dea26a64ffb9a3f72dc0866ce76160460f13b757bc4f0b4c8b","observation_id":"75ec9899-edcd-40e2-8855-d208388ede79","resolution":{"observed_at":"2026-05-17T13:17:43.060918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Taking advantage of their trust and naivety can be morally wrong","venue":null,"work_id":"ee973638-4320-4a52-b559-c6eb2dbd606b","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:12330ce020ec15a853700e0696124b1269268fd3a062a13da21eac28326d10a4","observation_id":"c4c5e115-b233-4864-9672-f126fbc4805a","resolution":{"observed_at":"2026-05-17T13:17:43.063957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Table 2: Examples of models’ responses to the English prompt on helpful and harmful evaluation","venue":null,"work_id":"26e7522f-734a-4dbc-8882-9b711e021686","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:e8d718392d7708703519be725049a60666ad228a0568472105cc902538c24ee7","observation_id":"7e6008d8-d889-45bf-993c-7f4a7593b6cc","resolution":{"observed_at":"2026-05-17T13:17:43.067928Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ae56fb5c-5c57-4756-83af-002f8f112b3f","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:02be10cc8a23384c98e0ba0182817facae9784fc13bf12daa812ca518ebaa811","observation_id":"00ca29f3-1af6-4287-b6d6-bfb0e72eb7bf","resolution":{"observed_at":"2026-05-17T13:17:43.071241Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"非法使用信息网络，视情节及后果表现，处5日以上10日以 下的行政拘留；情节严重、后果严重的，处10日以上15日以下的行政拘留","venue":null,"work_id":"6d39c8fe-66bc-4761-9e58-9b5c884e15e6","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:43863198aba4258b68a1c412b46d49a7c81b35865a24842d822faaf5355af1bf","observation_id":"41490a4e-bb89-416f-a14e-247a2c2e6e79","resolution":{"observed_at":"2026-05-17T13:17:43.074603Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"second hand of love","venue":null,"work_id":"b6fdd0ac-a592-4b2e-a510-9c845b42387f","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:bdba33d2b0c85f2dd6136cbaaa977c74699deeb7953969ec0fd0527b00f1a165","observation_id":"671038ff-fc7d-4189-b76b-a81cfb8df0c6","resolution":{"observed_at":"2026-05-17T13:17:43.079136Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Instead of rushing through life, take the time to savor the small things and appreciate the people around you","venue":null,"work_id":"ecb4bd88-3111-4f86-9d8a-03100fa2e5d4","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:7661deba2368c0ec9539f419cd67c79252bc68fdc04c61924c07fa43802b0393","observation_id":"b5b79e4d-b95f-4f83-94a0-8fef0be7e85d","resolution":{"observed_at":"2026-05-17T13:17:43.082149Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"When seen as a second hand of love, every moment counts, and we should make the most of them by being fully present and engaged in our relationships and experiences","venue":null,"work_id":"233552cb-b419-4ccf-a35c-4b963a96ddbc","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:10263902024321d970264021c76d23d0a0f424d69ae548f7a6d65f2409e3ca1b","observation_id":"42d25b89-75e8-401d-a937-79bed76d39a0","resolution":{"observed_at":"2026-05-17T13:17:43.085204Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We should focus on what truly matters to us and prioritize our time accordingly","venue":null,"work_id":"de7b86f7-6bb6-4deb-bed1-0838122fcaea","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:392a754ee122ea172b6a1415fb4a1ac036f68da22aa7eab21f1944f02600ce24","observation_id":"67bede84-c1a4-47b7-985c-2d82f0f1bd4b","resolution":{"observed_at":"2026-05-17T13:17:43.088537Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The Wandering Earth","venue":null,"work_id":"9425e7b1-dd75-4c86-aee8-127626ba815d","year":null},"citing_paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-17T13:17:42.848578Z"},"links":{"citing_paper":"/paper/2307.04964"},"observation_digest":"sha256:114adc24166f0eeb384102f3d62d469c0d9917159284515642d39b6a66b653e3","observation_id":"06c76c60-4bd7-43d7-8570-9f03c4d87e0f","resolution":{"observed_at":"2026-05-17T13:17:43.091413Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2307.04964","last_updated":"2023-07-18T08:44:47Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-11T01:55:24Z","title":"Secrets of RLHF in Large Language Models Part I: PPO"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":18,"verified_exact":14,"verified_fuzzy":27},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 29 inbound Pith citation observations for arXiv:2307.04964."}