{"as_of":"2026-08-09T06:05:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f2d5aac37e7884a0dbf54de5970d5a9367ae69709efebc8d84faec0097ae206b","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":37,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T04:52:40.534383Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":14,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2403.07691","last_updated":"2024-03-14T07:47:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-12T14:34:08Z","title":"ORPO: Monolithic Preference Optimization without Reference Model","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-16T09:34:04.394588Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2403.07691"},"observation_digest":"sha256:4d3fab745fb84111c0e3ca610a8437bb1deccf2f57a9a1e81e0c38fb75f411e9","observation_id":"b045bfd9-05d4-4f25-8c19-72a780e85c46","resolution":{"observed_at":"2026-05-16T09:34:04.769240Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2502.01456","last_updated":"2025-09-26T09:25:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-03T15:43:48Z","title":"Process Reinforcement through Implicit Rewards","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-11T20:23:30.763794Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2502.01456"},"observation_digest":"sha256:c789c53d9adfc47fc7dffc66d772171302c5d9edfa552d40b3f8fc6a9d7d2bd6","observation_id":"972de14a-91e6-49d4-94cb-6ccdaf5c574b","resolution":{"observed_at":"2026-05-11T20:23:30.944982Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-09T04:52:40.534383Z","title":"Bachmann, R., Kar, O","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.03429","last_updated":"2025-02-05T18:21:03Z","snapshot_observed_at":"2026-08-09T04:44:50.434091Z","submitted_at":"2025-02-05T18:21:03Z","title":"On Fairness of Unified Multimodal Large Language Model for Image Generation","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-09T04:52:40.534383Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2502.03429"},"observation_digest":"sha256:a6f16a0cbd0bb7ade8bc4c041f14e9221d1d80e6286a9414b3b854e543e79957","observation_id":"cf191c35-876d-4119-a643-6416b43f25a9","resolution":{"observed_at":"2026-08-09T04:52:40.534383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T15:07:59.145850Z","title":"A general theoretical paradigm to understand learning from human preferences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17134","last_updated":"2025-06-03T03:04:17Z","snapshot_observed_at":"2026-08-08T16:23:18.904775Z","submitted_at":"2025-05-22T04:05:02Z","title":"LongMagpie: A Self-synthesis Method for Generating Large-scale Long-context Instructions","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T15:07:59.145850Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2505.17134"},"observation_digest":"sha256:e9e232b2a66c1f2a85d2be056855525bce74a301432ad2c304fbbdc5751feb6d","observation_id":"b3153567-2340-453f-ad62-cc1375bc3e29","resolution":{"observed_at":"2026-08-07T15:07:59.145850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T14:31:45.997262Z","title":"A general theoretical paradigm to understand learning from human preferences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.21537","last_updated":"2025-05-24T09:07:13Z","snapshot_observed_at":"2026-08-08T02:37:32.325481Z","submitted_at":"2025-05-24T09:07:13Z","title":"OpenReview Should be Protected and Leveraged as a Community Asset for Research in the Era of Large Language Models","version":1},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-08-07T14:31:45.997262Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2505.21537"},"observation_digest":"sha256:23d7261ab9019896ce6b4f36173d549c5beca4d74ad2dd1bc51ca9e210fbabf9","observation_id":"54a2c4d8-ace3-40d0-979d-4e89e722de36","resolution":{"observed_at":"2026-08-07T14:31:45.997262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T12:43:47.923648Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23927","last_updated":"2025-05-29T18:22:02Z","snapshot_observed_at":"2026-08-09T00:47:06.977369Z","submitted_at":"2025-05-29T18:22:02Z","title":"Thompson Sampling in Online RLHF with General Function Approximation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:43:47.923648Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2505.23927"},"observation_digest":"sha256:96544e592f48e094dd547a35ce3d2bfcc77c1be71b05ee9e5e4f88038057f59e","observation_id":"f93d914f-0c4c-473a-82e8-2c013645e3ae","resolution":{"observed_at":"2026-08-07T12:43:47.923648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T10:57:42.228327Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03827","last_updated":"2025-06-04T10:57:18Z","snapshot_observed_at":"2026-08-08T01:31:46.984934Z","submitted_at":"2025-06-04T10:57:18Z","title":"Multi-objective Aligned Bidword Generation Model for E-commerce Search Advertising","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T10:57:42.228327Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2506.03827"},"observation_digest":"sha256:10be6401dc6a47488423a98a740e230da2f349c9dc10cce08789a4a6c8dc66b2","observation_id":"c9abefb6-8aaf-4792-96b9-8cd2dca89181","resolution":{"observed_at":"2026-08-07T10:57:42.228327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T05:22:07.333606Z","title":"Mohammad Gheshlaghi Azar, Mark Rowland, Bilal Piot, Daniel Guo, Daniele Calandriello, Michal Valko, and R´emi Munos","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08379","last_updated":"2025-06-10T02:43:47Z","snapshot_observed_at":"2026-08-08T17:37:44.974536Z","submitted_at":"2025-06-10T02:43:47Z","title":"Reinforce LLM Reasoning through Multi-Agent Reflection","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T05:22:07.333606Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2506.08379"},"observation_digest":"sha256:a80a9cab857231005c14d16efa023c276a8ba72bbbba964c5da574cf2613fb50","observation_id":"38a7ec8d-1f7a-4786-babf-f21180eb9c63","resolution":{"observed_at":"2026-08-07T05:22:07.333606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T00:16:06.358214Z","title":"Scaling direct preference optimization for fine-grained reward specification.arXiv preprint arXiv:2310.12036, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14903","last_updated":"2025-06-17T18:17:35Z","snapshot_observed_at":"2026-08-08T02:29:12.209103Z","submitted_at":"2025-06-17T18:17:35Z","title":"DETONATE: A Benchmark for Text-to-Image Alignment and Kernelized Direct Preference Optimization","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T00:16:06.358214Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2506.14903"},"observation_digest":"sha256:7a9fac85d59db59f82b18625d8f33600fa9401ecd7b6e38d2fb45b6a48d76cad","observation_id":"b2c59bc0-5896-4dd9-aef2-605091779e68","resolution":{"observed_at":"2026-08-07T00:16:06.358214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-06T19:18:40.283348Z","title":"Scaling laws for reward model overoptimization.arXiv preprint arXiv:2310.12036, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06013","last_updated":"2025-07-08T14:17:07Z","snapshot_observed_at":"2026-08-07T22:57:58.555368Z","submitted_at":"2025-07-08T14:17:07Z","title":"CogniSQL-R1-Zero: Lightweight Reinforced Reasoning for Efficient SQL Generation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T19:18:40.283348Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2507.06013"},"observation_digest":"sha256:e1b545217e275b2d1d589671504ba9842fa58def5d4056e3dff5d73b74d800f2","observation_id":"fe1636fe-aa28-4d0a-a15d-841dcc051593","resolution":{"observed_at":"2026-08-06T19:18:40.283348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2508.04149","last_updated":"2026-05-16T09:55:19Z","snapshot_observed_at":"2026-07-06T22:08:36.543090Z","submitted_at":"2025-08-06T07:24:14Z","title":"Difficulty-Based Preference Data Selection by DPO Implicit Reward Gap","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-21T23:46:24.208438Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2508.04149"},"observation_digest":"sha256:e557676b94fc520279f635bef917e41466514539be994c38b9b9265b84398c14","observation_id":"e1575096-4be4-46e7-b026-bd9c440d36d0","resolution":{"observed_at":"2026-05-21T23:50:47.607776Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T21:33:13.809159Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.08466","last_updated":"2025-08-11T20:53:37Z","snapshot_observed_at":"2026-08-05T21:33:06.047072Z","submitted_at":"2025-08-11T20:53:37Z","title":"Enhancing Small LLM Alignment through Margin-Based Objective Modifications under Resource Constraints","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T21:33:13.809159Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2508.08466"},"observation_digest":"sha256:e54580a2489e4381bd6bee8ce868e1e99a69e82aad0f61e22927b8db483c96b9","observation_id":"fec1937c-8fd6-4e10-b526-f01cd2d18c30","resolution":{"observed_at":"2026-08-05T21:33:13.809159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2509.20265","last_updated":"2026-04-29T12:32:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-24T15:52:36Z","title":"Failure Modes of Maximum Entropy RLHF","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-18T14:02:11.084514Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2509.20265"},"observation_digest":"sha256:417c049e1587942af6261344a01b7fdd8502db1a33e1964212624d2572a1111b","observation_id":"4e7e3696-975a-4dd5-9d5b-d36ed9ffd8d3","resolution":{"observed_at":"2026-05-18T14:02:39.813225Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-04T14:52:18.668425Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.22851","last_updated":"2026-07-02T18:42:52Z","snapshot_observed_at":"2026-08-08T05:03:31.962739Z","submitted_at":"2025-09-26T19:03:24Z","title":"Adaptive Margin RLHF via Preference over Preferences","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T14:52:18.668425Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2509.22851"},"observation_digest":"sha256:3ea0e362f1f5b0568bc337b4498507cc8fe0eb87b0463c7605023b08ed48f671","observation_id":"4deb9cf5-bfb4-4a4e-a885-a13706c67041","resolution":{"observed_at":"2026-08-04T14:52:18.668425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-03T14:24:19.502646Z","title":"A general theoretical paradigm to understand learning from human preferences, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2512.20806","last_updated":"2026-05-31T13:11:43Z","snapshot_observed_at":"2026-08-06T07:38:24.259000Z","submitted_at":"2025-12-23T22:13:14Z","title":"Safety Alignment of LMs via Non-cooperative Games","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-03T14:24:19.502646Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2512.20806"},"observation_digest":"sha256:9ab2ccf93b7bc8e9d1cb2c09904f3236373a634d10e87b5ee4d75c0d49b70c35","observation_id":"d058d0b3-5f42-4060-9235-c76e24e0b6db","resolution":{"observed_at":"2026-08-03T14:24:19.502646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.02626","last_updated":"2026-05-04T14:15:24Z","snapshot_observed_at":"2026-07-30T07:44:04.614445Z","submitted_at":"2026-05-04T14:15:24Z","title":"Gradient-Gated DPO: Stabilizing Preference Optimization in Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-08T18:35:13.659698Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.02626"},"observation_digest":"sha256:daeab29828746cf88d267931e7bef6995e25091e325f7c87172e29a02d65d4f0","observation_id":"03710aaf-d50c-4e56-b8a4-670bcc2e4707","resolution":{"observed_at":"2026-05-09T06:20:41.683804Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.06987","last_updated":"2026-05-07T22:05:23Z","snapshot_observed_at":"2026-07-31T05:30:22.249144Z","submitted_at":"2026-05-07T22:05:23Z","title":"Response Time Enhances Alignment with Heterogeneous Preferences","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-11T01:04:26.288913Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.06987"},"observation_digest":"sha256:b7a8e8eee8169be2bd6027c11cff37afd73d7c0e1be612eefcb178079e0f696c","observation_id":"ae04bdcb-ef5f-4f03-9cd2-af5fee1a1754","resolution":{"observed_at":"2026-05-11T01:05:50.004199Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.11726","last_updated":"2026-05-13T15:38:02Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T08:09:42Z","title":"Block-R1: Rethinking the Role of Block Size in Multi-domain Reinforcement Learning for Diffusion Large Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T07:03:00.503644Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.11726"},"observation_digest":"sha256:58fa7ca4459534bbdd87ed871d37f1fc7f71b0a7326f64b28829bfa4ddd41cc0","observation_id":"2deb40d5-6e9a-486a-9ba4-13dc4827f5c1","resolution":{"observed_at":"2026-05-13T07:07:28.007161Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.11726","last_updated":"2026-05-13T15:38:02Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T08:09:42Z","title":"Block-R1: Rethinking the Role of Block Size in Multi-domain Reinforcement Learning for Diffusion Large Language Models","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-14T21:06:01.667173Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.11726"},"observation_digest":"sha256:bc3b96cb130e9b494227c7fd10037e18ea58926a1675bc3a8538cb892a3b514b","observation_id":"36ad412e-a310-41da-b084-7a7d3f8d7d7d","resolution":{"observed_at":"2026-05-14T21:19:28.313009Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.11906","last_updated":"2026-05-12T10:18:49Z","snapshot_observed_at":"2026-07-31T22:11:05.515608Z","submitted_at":"2026-05-12T10:18:49Z","title":"YFPO: A Preliminary Study of Yoked Feature Preference Optimization with Neuron-Guided Rewards for Mathematical Reasoning","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-13T05:29:42.381280Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.11906"},"observation_digest":"sha256:5c5d40f8928010dd2b461f66c267ea3136846ed25fd16cef547ee6973b6f600f","observation_id":"89c015fc-013f-4f68-85d5-8f127abb6416","resolution":{"observed_at":"2026-05-13T05:32:19.327417Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.12288","last_updated":"2026-06-10T07:32:21Z","snapshot_observed_at":"2026-07-06T23:23:59.123377Z","submitted_at":"2026-05-12T15:44:33Z","title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","version":1},"reference_index":136,"source":"arxiv_source","source_observed_at":"2026-05-13T04:55:55.013900Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.12288"},"observation_digest":"sha256:4c692eddba6fb94ea1b6a9ecd93c650e7020bac20108b8a3354d08b7487a86aa","observation_id":"4b7e7949-a768-47e2-85a4-642bbdd8a483","resolution":{"observed_at":"2026-05-13T04:57:17.181807Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.12288","last_updated":"2026-06-10T07:32:21Z","snapshot_observed_at":"2026-07-06T23:23:59.123377Z","submitted_at":"2026-05-12T15:44:33Z","title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","version":2},"reference_index":136,"source":"arxiv_source","source_observed_at":"2026-05-15T05:41:10.714594Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.12288"},"observation_digest":"sha256:a17732f63dde01345666d7319ffb6fc78222679d06e1fd13f2087eec02db7e8d","observation_id":"09a58498-7564-4110-a7c0-603eedba08a4","resolution":{"observed_at":"2026-05-15T05:45:06.641433Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.21854","last_updated":"2026-06-06T09:58:33Z","snapshot_observed_at":"2026-08-02T07:58:30.433073Z","submitted_at":"2026-05-21T01:02:41Z","title":"CrossVLA: Cross-Paradigm Post-Training and Inference Optimization for Vision-Language-Action Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-22T08:07:51.697353Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.21854"},"observation_digest":"sha256:63d7c91894857f0900ad0d93f623695c790d7d5b3d5f9376e219854e883e354e","observation_id":"ce4112a6-da47-454b-a8cf-6dbba167bf8f","resolution":{"observed_at":"2026-05-22T08:11:17.214389Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.21854","last_updated":"2026-06-06T09:58:33Z","snapshot_observed_at":"2026-08-02T07:58:30.433073Z","submitted_at":"2026-05-21T01:02:41Z","title":"CrossVLA: Cross-Paradigm Post-Training and Inference Optimization for Vision-Language-Action Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T17:48:02.228941Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.21854"},"observation_digest":"sha256:0d8bb5ed34fa5e07e437ff1531a31964654324f50fc9c1c23977a2567b6d4b9d","observation_id":"735f9d5d-4baf-4dcc-be85-c42485d78836","resolution":{"observed_at":"2026-07-01T15:05:47.998440Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2605.28440","last_updated":"2026-05-27T13:05:49Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T13:05:49Z","title":"AdaDPO: Self-Adaptive Direct Preference Optimization with Balanced Gradient Updates","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T12:29:55.729913Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2605.28440"},"observation_digest":"sha256:721cea93be9fcb1e4f713ef5ec07de83577953cb3118ab02a8d5206a2e70e9de","observation_id":"5b41e8e6-8bf3-4891-842a-8fd6f1a8a774","resolution":{"observed_at":"2026-06-29T12:33:24.402718Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.03089","last_updated":"2026-06-16T07:22:07Z","snapshot_observed_at":"2026-07-06T23:43:22.513369Z","submitted_at":"2026-06-02T03:17:56Z","title":"Constitutional On-Policy Safe Distillation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T11:47:14.135793Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.03089"},"observation_digest":"sha256:39347b6f333e0d609c939670ad696471233b0461c639a8117df219b3a59d910b","observation_id":"bb7add3f-651c-4f56-bc78-9cd92d032cec","resolution":{"observed_at":"2026-07-02T01:36:25.512570Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.05468","last_updated":"2026-06-03T21:47:43Z","snapshot_observed_at":"2026-07-06T23:45:28.379468Z","submitted_at":"2026-06-03T21:47:43Z","title":"FlowPRO: Reward-Free Reinforced Fine-Tuning of Flow-Matching VLAs via Proximalized Preference Optimization","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-28T05:38:11.089753Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.05468"},"observation_digest":"sha256:772332e969ec2ee496029ad340b77ed879f7b89e224ba69b199be611984b6e35","observation_id":"75c04942-f7a4-49e3-81b1-6bda7cfbccb3","resolution":{"observed_at":"2026-07-02T09:06:49.361855Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.23740","last_updated":"2026-06-21T15:34:21Z","snapshot_observed_at":"2026-08-02T05:47:42.942051Z","submitted_at":"2026-06-21T15:34:21Z","title":"Weight-Space Geometry of Offline Reasoning Training","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-06-26T10:26:28.702213Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.23740"},"observation_digest":"sha256:a73a393d94c39dfe3a4c5ef10393c291d5c6ef0e3a5a7446c2c58748e55513aa","observation_id":"5b049f08-6e51-467c-9c92-342a0f7eb1a2","resolution":{"observed_at":"2026-07-04T09:09:43.385341Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.24004","last_updated":"2026-06-29T07:01:12Z","snapshot_observed_at":"2026-08-02T11:15:29.427964Z","submitted_at":"2026-06-22T23:21:55Z","title":"Towards Spec Learning: Inference-Time Alignment from Preference Pairs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-26T07:49:36.816100Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.24004"},"observation_digest":"sha256:535933acb0fd32e91a0eef6188f8e669fba86836c8d6997f6442ece22fb4838e","observation_id":"a11da4f5-e823-4ed4-a425-56b5519e8320","resolution":{"observed_at":"2026-07-04T11:39:47.161717Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.24004","last_updated":"2026-06-29T07:01:12Z","snapshot_observed_at":"2026-08-02T11:15:29.427964Z","submitted_at":"2026-06-22T23:21:55Z","title":"Towards Spec Learning: Inference-Time Alignment from Preference Pairs","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T10:17:33.176525Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.24004"},"observation_digest":"sha256:6d8d86acd404659ca83a1865816664ea005fcda3c8181810b42d1ce6d5171245","observation_id":"38364724-f099-488a-a77c-71f5162f8b76","resolution":{"observed_at":"2026-06-30T12:44:39.556729Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":"2310.12036","doi":"10.48550/arxiv.2310.12036","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A general theoretical paradigm to understand learning from human preferences.arXiv preprint arXiv:2310.12036","venue":"arXiv (Cornell University)","work_id":"44673d8e-2cc2-4818-86d3-24bc812aa41c","year":2023},"citing_paper":{"arxiv_id":"2606.24937","last_updated":"2026-07-27T15:17:17Z","snapshot_observed_at":"2026-08-02T23:19:25.465662Z","submitted_at":"2026-06-22T17:48:54Z","title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-26T08:09:57.542558Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.24937"},"observation_digest":"sha256:2836deae9e2ff416d30e39017237363ca5b18e4c8f4d25ed6ff7f8e93b1994f6","observation_id":"a1437ddd-5690-4ed6-a8bc-fa94eab0e2ef","resolution":{"observed_at":"2026-07-04T11:09:46.374927Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T21:08:08.42545+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.42545+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-02T10:27:16.169555Z","title":"A General Theoretical Paradigm to Understand Learning from Human Feedback.arXiv Preprint arXiv:2310.12036, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.24937","last_updated":"2026-07-27T15:17:17Z","snapshot_observed_at":"2026-08-02T23:19:25.465662Z","submitted_at":"2026-06-22T17:48:54Z","title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T10:27:16.169555Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2606.24937"},"observation_digest":"sha256:cb17e536f809f1394bdc1f757b68d2e48b01165876168dc9f2e354ca706508bc","observation_id":"b6fc0ef8-98a3-4acf-9eb1-8386c9e100f9","resolution":{"observed_at":"2026-08-02T10:27:16.169555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-07-12T04:43:45.592808Z","title":"arXiv preprint arXiv:2310.12036 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03126","last_updated":"2026-07-07T03:50:51Z","snapshot_observed_at":"2026-08-02T17:17:56.686166Z","submitted_at":"2026-07-03T09:14:27Z","title":"ACPO: Adaptive Credit Policy Optimization via Fine-Grained Surrogate Entropy","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-07-12T04:43:45.592808Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2607.03126"},"observation_digest":"sha256:8730a0a67fec72a8b47b0125e88e1333773538dec138a8b54e79e91b6820a794","observation_id":"a5e8bc7a-5eba-4e50-a3b7-20421d5bdd65","resolution":{"observed_at":"2026-07-12T04:43:45.592808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-07-11T13:53:36.775836Z","title":"arXiv preprint arXiv:2310.12036 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-02T10:24:43.977557Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":1},"reference_index":163,"source":"arxiv_source","source_observed_at":"2026-07-11T13:53:36.775836Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:f61b1a8c69eaa6d45688e0feaf31495a7428fda1d4f9b7d949c3ce2b343bf533","observation_id":"a8fe755a-582d-4e88-b4ca-db7b229fe51b","resolution":{"observed_at":"2026-07-11T13:53:36.775836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-02T08:40:51.116483Z","title":"arXiv preprint arXiv:2310.12036 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-02T10:24:43.977557Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":3},"reference_index":164,"source":"arxiv_source","source_observed_at":"2026-08-02T08:40:51.116483Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:4977c880680c81180dc91ea15ade6db180dad7c166d59a2452f1df30ef392cda","observation_id":"b839b03e-a437-4a0d-a884-49efdc099734","resolution":{"observed_at":"2026-08-02T08:40:51.116483Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-07-31T12:20:07.021635Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.28282","last_updated":"2026-07-30T14:31:07Z","snapshot_observed_at":"2026-08-06T16:34:19.905250Z","submitted_at":"2026-07-30T14:31:07Z","title":"(Towards) Scalable Reliable Automated Evaluation with Large Language Models","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-07-31T12:20:07.021635Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2607.28282"},"observation_digest":"sha256:26df775cfe9104ff5c51668d25137c0feb802a8d71a3e71e5f17f2b1aef08ae7","observation_id":"8899f132-d89f-4e05-bac8-94441a09aef5","resolution":{"observed_at":"2026-07-31T12:20:07.021635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-07T00:13:50.414984Z","title":"A general theoretical paradigm to understand learning from human preferences","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02713","last_updated":"2026-08-03T17:59:58Z","snapshot_observed_at":"2026-08-07T23:09:56.055766Z","submitted_at":"2026-08-03T17:59:58Z","title":"Quo Vadis, World Modeling?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T00:13:50.414984Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2608.02713"},"observation_digest":"sha256:11be1dbf7d442dac67fb4862912cf32ccbd07c769a17c8f86de4b7e569b625cb","observation_id":"a51d4882-beae-4871-b73f-73d1a086b0f0","resolution":{"observed_at":"2026-08-07T00:13:50.414984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.12036/citation-record","integrity":"/paper/2310.12036/integrity","json":"/paper/2310.12036/citation-record.json","paper":"/paper/2310.12036"},"outbound":[],"paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-07-06T16:35:07.121479Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 37 inbound Pith citation observations for arXiv:2310.12036."}