{"as_of":"2026-08-08T17:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6556f9d84dd1873e74177b1f87e7524db5fbe87370ef72bdc35e256d86308017","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":10,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":10,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":10,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":10,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:08:05.178661Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":2,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.13639","last_updated":"2024-04-30T14:36:26Z","snapshot_observed_at":"2026-08-05T15:32:38.926507Z","submitted_at":"2023-10-20T16:37:56Z","title":"Contrastive Preference Learning: Learning from Human Feedback without RL","version":3},"cited_work":{"arxiv_id":"2310.13639","doi":"10.48550/arxiv.2310.13639","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.13639","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Contrastive prefence learning: Learning from human feedback without rl","venue":"arXiv (Cornell University)","work_id":"d302bae9-7d89-4fe5-a0e8-e81576771123","year":2024},"citing_paper":{"arxiv_id":"2411.00361","last_updated":"2026-04-15T18:16:21Z","snapshot_observed_at":"2026-07-06T19:43:23.653382Z","submitted_at":"2024-11-01T04:58:40Z","title":"Direct Preference Optimization for Primitive-Enabled Hierarchical RL: A Bilevel Approach","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-23T18:30:38.818711Z"},"links":{"cited_paper":"/paper/2310.13639","citing_paper":"/paper/2411.00361"},"observation_digest":"sha256:396ab9d3e9ae4b525b9ca24e74af2d434644fdd995c68853deb8ef1e77e7e6ba","observation_id":"db549edc-196a-4039-b3d5-0c454212e1a4","resolution":{"observed_at":"2026-05-23T18:33:19.704972Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13639","last_updated":"2024-04-30T14:36:26Z","snapshot_observed_at":"2026-08-05T15:32:38.926507Z","submitted_at":"2023-10-20T16:37:56Z","title":"Contrastive Preference Learning: Learning from Human Feedback without RL","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.13639","snapshot_observed_at":"2026-08-07T05:08:05.178661Z","title":"Bradley Knox, and Dorsa Sadigh","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08998","last_updated":"2025-06-10T17:17:48Z","snapshot_observed_at":"2026-08-07T04:54:47.198895Z","submitted_at":"2025-06-10T17:17:48Z","title":"On Monotonicity in AI Alignment","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T05:08:05.178661Z"},"links":{"cited_paper":"/paper/2310.13639","citing_paper":"/paper/2506.08998"},"observation_digest":"sha256:87cd9745742098196d7cf8f2fff52b3467413bf137d6898802287ede19abb261","observation_id":"331e7915-e8e4-4789-b9a3-9d870694a02d","resolution":{"observed_at":"2026-08-07T05:08:05.178661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13639","last_updated":"2024-04-30T14:36:26Z","snapshot_observed_at":"2026-08-05T15:32:38.926507Z","submitted_at":"2023-10-20T16:37:56Z","title":"Contrastive Preference Learning: Learning from Human Feedback without RL","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.13639","snapshot_observed_at":"2026-08-07T05:02:05.531884Z","title":"B., and Sadigh, D","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09176","last_updated":"2025-06-10T18:43:26Z","snapshot_observed_at":"2026-08-08T08:46:07.979494Z","submitted_at":"2025-06-10T18:43:26Z","title":"Robot-Gated Interactive Imitation Learning with Adaptive Intervention Mechanism","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-07T05:02:05.531884Z"},"links":{"cited_paper":"/paper/2310.13639","citing_paper":"/paper/2506.09176"},"observation_digest":"sha256:f90362305b78df78dc4f918992ad69a1c44d1b32ab503d9567ead07bc1faccb5","observation_id":"ba3d006b-be9a-4b58-923b-160fdf816342","resolution":{"observed_at":"2026-08-07T05:02:05.531884Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13639","last_updated":"2024-04-30T14:36:26Z","snapshot_observed_at":"2026-08-05T15:32:38.926507Z","submitted_at":"2023-10-20T16:37:56Z","title":"Contrastive Preference Learning: Learning from Human Feedback without RL","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.13639","snapshot_observed_at":"2026-08-06T16:34:25.031856Z","title":"Contrastive prefence learning: Learning from human feedback without rl.arXiv preprint arXiv:2310.13639,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.13158","last_updated":"2025-07-17T14:22:24Z","snapshot_observed_at":"2026-08-07T07:16:46.127762Z","submitted_at":"2025-07-17T14:22:24Z","title":"Inverse Reinforcement Learning Meets Large Language Model Post-Training: Basics, Advances, and Opportunities","version":1},"reference_index":1994,"source":"pdf_text","source_observed_at":"2026-08-06T16:34:25.031856Z"},"links":{"cited_paper":"/paper/2310.13639","citing_paper":"/paper/2507.13158"},"observation_digest":"sha256:b546a49b4fbaeb1d5319e0e5ff357400aeeebb8cb99f6b80d15224b0e8647f3c","observation_id":"2be71f98-e9ce-498d-9c6c-6f9980561ab2","resolution":{"observed_at":"2026-08-06T16:34:25.031856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13639","last_updated":"2024-04-30T14:36:26Z","snapshot_observed_at":"2026-08-05T15:32:38.926507Z","submitted_at":"2023-10-20T16:37:56Z","title":"Contrastive Preference Learning: Learning from Human Feedback without RL","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.13639","snapshot_observed_at":"2026-08-04T09:50:04.047978Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.13434","last_updated":"2026-07-28T12:08:40Z","snapshot_observed_at":"2026-08-08T01:32:59.519778Z","submitted_at":"2025-10-15T11:30:49Z","title":"$M^2PO$: Multi-Perspective Multi-Pair Preference Optimization for Machine Translation","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-04T09:50:04.047978Z"},"links":{"cited_paper":"/paper/2310.13639","citing_paper":"/paper/2510.13434"},"observation_digest":"sha256:c331b223832626b96ded9e8d7344a8296993a29f8e107eadc712a2160970f784","observation_id":"cea6677b-c46f-4e5a-bab4-70b51806097d","resolution":{"observed_at":"2026-08-04T09:50:04.047978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13639","last_updated":"2024-04-30T14:36:26Z","snapshot_observed_at":"2026-08-05T15:32:38.926507Z","submitted_at":"2023-10-20T16:37:56Z","title":"Contrastive Preference Learning: Learning from Human Feedback without RL","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.13639","snapshot_observed_at":"2026-08-03T05:49:24.586560Z","title":"Contrastive preference learning: learning from human feedback without rl.arXiv preprint arXiv:2310.13639,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.01357","last_updated":"2026-06-08T05:19:57Z","snapshot_observed_at":"2026-08-08T01:13:19.346639Z","submitted_at":"2026-02-01T17:50:59Z","title":"Your Self-Play Algorithm is Secretly an Adversarial Imitator: Understanding LLM Self-Play through the Lens of Imitation Learning","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-03T05:49:24.586560Z"},"links":{"cited_paper":"/paper/2310.13639","citing_paper":"/paper/2602.01357"},"observation_digest":"sha256:053c1254f22ff94d5726ede57f4cd1a6a74843ed6fe7720c63c31820584e0898","observation_id":"dd8b1b72-f44e-4bb1-b20a-643e8a72ff31","resolution":{"observed_at":"2026-08-03T05:49:24.586560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13639","last_updated":"2024-04-30T14:36:26Z","snapshot_observed_at":"2026-08-05T15:32:38.926507Z","submitted_at":"2023-10-20T16:37:56Z","title":"Contrastive Preference Learning: Learning from Human Feedback without RL","version":3},"cited_work":{"arxiv_id":"2310.13639","doi":"10.48550/arxiv.2310.13639","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.13639","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Contrastive prefence learning: Learning from human feedback without rl","venue":"arXiv (Cornell University)","work_id":"d302bae9-7d89-4fe5-a0e8-e81576771123","year":2024},"citing_paper":{"arxiv_id":"2606.09124","last_updated":"2026-06-08T07:18:44Z","snapshot_observed_at":"2026-08-04T01:46:39.439263Z","submitted_at":"2026-06-08T07:18:44Z","title":"A Regret Minimization Framework on Preference Learning in Large Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T16:34:19.323865Z"},"links":{"cited_paper":"/paper/2310.13639","citing_paper":"/paper/2606.09124"},"observation_digest":"sha256:eec5b55e30531e3af3986ad53ea5967b7abdca271c51ba9f84b4dd534a2b7ec4","observation_id":"960af314-a171-4893-b5e7-fac0da6c4744","resolution":{"observed_at":"2026-07-03T01:27:30.773321Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13639","last_updated":"2024-04-30T14:36:26Z","snapshot_observed_at":"2026-08-05T15:32:38.926507Z","submitted_at":"2023-10-20T16:37:56Z","title":"Contrastive Preference Learning: Learning from Human Feedback without RL","version":3},"cited_work":{"arxiv_id":"2310.13639","doi":"10.48550/arxiv.2310.13639","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.13639","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Contrastive prefence learning: Learning from human feedback without rl","venue":"arXiv (Cornell University)","work_id":"d302bae9-7d89-4fe5-a0e8-e81576771123","year":2024},"citing_paper":{"arxiv_id":"2606.12372","last_updated":"2026-06-10T17:38:24Z","snapshot_observed_at":"2026-08-06T11:07:11.202179Z","submitted_at":"2026-06-10T17:38:24Z","title":"UniIntervene: Agentic Intervention for Efficient Real-World Reinforcement Learning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-27T09:46:59.746745Z"},"links":{"cited_paper":"/paper/2310.13639","citing_paper":"/paper/2606.12372"},"observation_digest":"sha256:e77013c4f98b6241a821eaf172b5a8559acb60864eed726bd0e1514021db876c","observation_id":"42ec9632-a473-4111-bb90-6f7d88dd068c","resolution":{"observed_at":"2026-07-03T10:58:02.723259Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13639","last_updated":"2024-04-30T14:36:26Z","snapshot_observed_at":"2026-08-05T15:32:38.926507Z","submitted_at":"2023-10-20T16:37:56Z","title":"Contrastive Preference Learning: Learning from Human Feedback without RL","version":3},"cited_work":{"arxiv_id":"2310.13639","doi":"10.48550/arxiv.2310.13639","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.13639","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Contrastive prefence learning: Learning from human feedback without rl","venue":"arXiv (Cornell University)","work_id":"d302bae9-7d89-4fe5-a0e8-e81576771123","year":2024},"citing_paper":{"arxiv_id":"2606.18703","last_updated":"2026-06-17T05:30:47Z","snapshot_observed_at":"2026-08-07T03:42:52.919517Z","submitted_at":"2026-06-17T05:30:47Z","title":"Contextualizing Biological Language Models across Modalities via Logit-Space Contrastive Alignment","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-06-26T21:31:07.417680Z"},"links":{"cited_paper":"/paper/2310.13639","citing_paper":"/paper/2606.18703"},"observation_digest":"sha256:16a19f10b2b07cfc5997dce4ad7927136303994c35462e5d313ab87591ab2b55","observation_id":"9f84ff1c-b804-45e5-a04b-554849bd0ed0","resolution":{"observed_at":"2026-06-26T21:40:08.481236Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13639","last_updated":"2024-04-30T14:36:26Z","snapshot_observed_at":"2026-08-05T15:32:38.926507Z","submitted_at":"2023-10-20T16:37:56Z","title":"Contrastive Preference Learning: Learning from Human Feedback without RL","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.13639","snapshot_observed_at":"2026-08-02T13:58:42.282065Z","title":"Contrastive preference learning: learning from human feedback without rl","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.18258","last_updated":"2026-05-15T09:09:57Z","snapshot_observed_at":"2026-08-07T01:09:03.575259Z","submitted_at":"2026-05-15T09:09:57Z","title":"S2T-RLHF: Hierarchical Credit Assignment for Stable Preference-Based RLHF","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-02T13:58:42.282065Z"},"links":{"cited_paper":"/paper/2310.13639","citing_paper":"/paper/2607.18258"},"observation_digest":"sha256:208a14c686e8d3bf8b6302e20397dbea44c785fe0b7d96277fd945bdf1c79d7a","observation_id":"382bfbfe-7516-4062-ae85-35d2c6448ef3","resolution":{"observed_at":"2026-08-02T13:58:42.282065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.13639/citation-record","integrity":"/paper/2310.13639/integrity","json":"/paper/2310.13639/citation-record.json","paper":"/paper/2310.13639"},"outbound":[],"paper":{"arxiv_id":"2310.13639","last_updated":"2024-04-30T14:36:26Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-05T15:32:38.926507Z","submitted_at":"2023-10-20T16:37:56Z","title":"Contrastive Preference Learning: Learning from Human Feedback without RL"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 10 inbound Pith citation observations for arXiv:2310.13639."}