{"as_of":"2026-08-13T21:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3a8fbc8fcdc661167955f500b7cf5430b8fa52ef4842528e21841016fceb5b23","coverage":[{"denominator":26,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T12:41:20.743435Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.15282/citation-record","integrity":"/paper/2412.15282/integrity","json":"/paper/2412.15282/citation-record.json","paper":"/paper/2412.15282"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-11T12:41:20.670025Z","title":"The llama 3 herd of models.CoRR, abs/2407.21783,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.670025Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:b492e92469ebaf0366d52d47b22fd76f4a937182f52acd7a0a3f767b6206a2fc","observation_id":"44a89a05-35e4-4137-b593-d08edd9d19ed","resolution":{"observed_at":"2026-08-11T12:41:20.670025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-11T12:41:20.676408Z","title":"Brown, Jack Clark, Sam McCandlish, Chris Olah, Benjamin Mann, and Jared Kaplan","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.676408Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:07409432ed616a184b63b6fa5fa0733c225905bec224eb9125dbe3d1076316de","observation_id":"e49292da-55ef-49b5-96b3-85483da3f51f","resolution":{"observed_at":"2026-08-11T12:41:20.676408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09279","last_updated":"2024-10-07T21:24:59Z","snapshot_observed_at":"2026-08-12T23:43:26.084884Z","submitted_at":"2024-06-13T16:17:21Z","title":"Unpacking DPO and PPO: Disentangling Best Practices for Learning from Preference Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09279","snapshot_observed_at":"2026-08-11T12:41:20.689532Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.689532Z"},"links":{"cited_paper":"/paper/2406.09279","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:1ac9966164049139d3207297f8c01c3209ad93e46bd610ed8ef63c5d761da288","observation_id":"d803d4db-5551-45f4-a5c0-be8fa0309680","resolution":{"observed_at":"2026-08-11T12:41:20.689532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.14774","last_updated":"2025-05-28T19:21:23Z","snapshot_observed_at":"2026-08-12T22:56:55.030568Z","submitted_at":"2024-08-27T04:31:58Z","title":"Instruct-SkillMix: A Powerful Pipeline for LLM Instruction Tuning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.14774","snapshot_observed_at":"2026-08-11T12:41:20.692264Z","title":"Saeed Khaki, JinJin Li, Lan Ma, Liu Yang, and Prathap Ramachandra","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.692264Z"},"links":{"cited_paper":"/paper/2408.14774","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:8b0d5e637f91e752794746deac04ddc98d89f993fb3d07a994a9c65c46dc6318","observation_id":"f8c7ddc2-4d30-4df0-a18e-36bbb5d2d883","resolution":{"observed_at":"2026-08-11T12:41:20.692264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.18629","last_updated":"2024-06-26T17:43:06Z","snapshot_observed_at":"2026-08-06T00:24:52.274888Z","submitted_at":"2024-06-26T17:43:06Z","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.18629","snapshot_observed_at":"2026-08-11T12:41:20.695300Z","title":"https://doi.org/10.18653/v1/2024.findings-naacl.108","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.695300Z"},"links":{"cited_paper":"/paper/2406.18629","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:9dfd1be086e5987d67203a1dd082c7624c51101dfc7096c28d6802cd05013c54","observation_id":"37ad7b07-5d02-43ae-a588-8ce714176bf7","resolution":{"observed_at":"2026-08-11T12:41:20.695300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.18629","last_updated":"2024-06-26T17:43:06Z","snapshot_observed_at":"2026-08-06T00:24:52.274888Z","submitted_at":"2024-06-26T17:43:06Z","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.18629","snapshot_observed_at":"2026-08-11T12:41:20.698420Z","title":"https://doi.org/10.48550/arXiv.2406.18629","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.698420Z"},"links":{"cited_paper":"/paper/2406.18629","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:891a21da10365c42fdd75e211e19391b3a1281eb11109957e2b40a10cdde5860","observation_id":"d6784637-1a2b-4c90-840f-60079620f06e","resolution":{"observed_at":"2026-08-11T12:41:20.698420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-11T12:41:20.700856Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.700856Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:68907df7446cc4273fb501cf05a8d240abf16c4bd0ca820555984b595e6298d8","observation_id":"aa0feaca-9e16-42a4-91b4-f2f712e84369","resolution":{"observed_at":"2026-08-11T12:41:20.700856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-11T12:41:20.703382Z","title":"Long Ouyang, Jeffrey Wu, Xu Jiang, Diogo Almeida, Carroll L","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.703382Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:2b757eb3ea2631157bc71bc33bae2c67cdfdfd4f5438c4bcd8e85cfd07f5187e","observation_id":"f482485d-3b53-4084-a950-40d7f860e247","resolution":{"observed_at":"2026-08-11T12:41:20.703382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19733","last_updated":"2024-06-26T01:28:35Z","snapshot_observed_at":"2026-08-13T03:45:12.203086Z","submitted_at":"2024-04-30T17:28:05Z","title":"Iterative Reasoning Preference Optimization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19733","snapshot_observed_at":"2026-08-11T12:41:20.706243Z","title":"Richard Yuanzhe Pang, Weizhe Yuan, Kyunghyun Cho, He He, Sainbayar Sukhbaatar, and Jason Weston","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.706243Z"},"links":{"cited_paper":"/paper/2404.19733","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:67c1d140b9ef613206c8149614e5c6fa1eb8266b5b5c7a3a04235fbf005c4a24","observation_id":"1af80b31-0e51-4f03-909c-6d85ce533158","resolution":{"observed_at":"2026-08-11T12:41:20.706243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19733","last_updated":"2024-06-26T01:28:35Z","snapshot_observed_at":"2026-08-13T03:45:12.203086Z","submitted_at":"2024-04-30T17:28:05Z","title":"Iterative Reasoning Preference Optimization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19733","snapshot_observed_at":"2026-08-11T12:41:20.709377Z","title":"https: //doi.org/10.48550/arXiv.2404.19733","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.709377Z"},"links":{"cited_paper":"/paper/2404.19733","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:62bf958f96e8b16e00d23156bb4f71e671582233bfadfe87d3bca9eae27fc3c7","observation_id":"c20e5985-e0d4-442f-81a9-344753191cdf","resolution":{"observed_at":"2026-08-11T12:41:20.709377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-11T12:41:20.720234Z","title":"Xuezhi Wang, Jason Wei, Dale Schuurmans, Quoc Le, Ed Huai hsin Chi, and Denny Zhou","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.720234Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:caacdfe29489020e9494a000138ecd84fe8ecfb10c619c25dcc9175c16afbe58","observation_id":"3e61e3d9-d6d9-48f1-9f7e-b6369487c1ab","resolution":{"observed_at":"2026-08-11T12:41:20.720234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:41:20.907117Z","title":"Yizhong Wang, Yeganeh Kordi, Swaroop Mishra, Alisa Liu, Noah A","venue":null,"work_id":"a2691a25-c3b8-46b4-9489-598964a6e3a0","year":2023},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.722980Z"},"links":{"citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:31eec14f128c3daf2b966ce993dc20f12c4a2b83999fd87b5f51bf5b72a7cf9f","observation_id":"64414039-2874-4c9c-8f3c-5600256de41e","resolution":{"observed_at":"2026-08-11T12:41:20.911248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.03978","last_updated":"2024-10-31T08:11:04Z","snapshot_observed_at":"2026-08-12T23:28:45.822390Z","submitted_at":"2024-07-04T14:50:45Z","title":"Benchmarking Complex Instruction-Following with Multiple Constraints Composition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.03978","snapshot_observed_at":"2026-08-11T12:41:20.725094Z","title":"https://doi.org/10.18653/v1/2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.725094Z"},"links":{"cited_paper":"/paper/2407.03978","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:a739b8e65f5664c4e0f8ea893297e55a4648b1596c29e471950dba64acbab53d","observation_id":"afd7252f-1140-4724-b201-b1ce7948a376","resolution":{"observed_at":"2026-08-11T12:41:20.725094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.03978","last_updated":"2024-10-31T08:11:04Z","snapshot_observed_at":"2026-08-12T23:28:45.822390Z","submitted_at":"2024-07-04T14:50:45Z","title":"Benchmarking Complex Instruction-Following with Multiple Constraints Composition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.03978","snapshot_observed_at":"2026-08-11T12:41:20.727453Z","title":"https://doi.org/10.48550/arXiv.2407.03978","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.727453Z"},"links":{"cited_paper":"/paper/2407.03978","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:05b0e217cfd9634c3b64bfa7ba51135eca489b89234d13bd41c44f408a437659","observation_id":"95259ad0-a5c4-45cc-bcae-a6225943434e","resolution":{"observed_at":"2026-08-11T12:41:20.727453Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.15595","last_updated":"2026-06-09T05:48:30Z","snapshot_observed_at":"2026-08-12T22:19:01.850354Z","submitted_at":"2024-10-21T02:27:24Z","title":"A Comprehensive Survey of Direct Preference Optimization: Datasets, Theories, Variants, and Applications","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.15595","snapshot_observed_at":"2026-08-11T12:41:20.729703Z","title":"Yuxi Xie, Anirudh Goyal, Wenyue Zheng, Min-Yen Kan, Timothy P","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.729703Z"},"links":{"cited_paper":"/paper/2410.15595","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:a7fc4932d0b520dbe46ed508beea4e72ee141a5ef08bdb8751a59716fc83e2f1","observation_id":"f028a299-d28d-459f-84e3-0fa6a25f4242","resolution":{"observed_at":"2026-08-11T12:41:20.729703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-13T00:17:43.021579Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-11T12:41:20.732265Z","title":"https://doi.org/10.48550/arXiv.2405.00451","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.732265Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:c4b944d0e8d97665aa076f9d8ea417f687f267dcdfc8de1d13fb9d73d3b7af9a","observation_id":"9e51b3f5-5634-463f-bc0e-2fa5f74376bd","resolution":{"observed_at":"2026-08-11T12:41:20.732265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.16682","last_updated":"2024-04-22T22:51:32Z","snapshot_observed_at":"2026-08-13T04:53:15.268224Z","submitted_at":"2023-12-27T18:53:09Z","title":"Some things are more CRINGE than others: Iterative Preference Optimization with the Pairwise Cringe Loss","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.16682","snapshot_observed_at":"2026-08-11T12:41:20.735095Z","title":"https://doi.org/10.48550/arXiv.2312.16682","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.735095Z"},"links":{"cited_paper":"/paper/2312.16682","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:d8e27a10637a99e7d2ad0b9a55a1bfc5aa4e683705fb99eb736c7ac814ddfc06","observation_id":"20a3b165-059b-4ef6-ae1c-2697585c1088","resolution":{"observed_at":"2026-08-11T12:41:20.735095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.03816","last_updated":"2024-11-18T05:36:16Z","snapshot_observed_at":"2026-08-12T23:49:03.076666Z","submitted_at":"2024-06-06T07:40:00Z","title":"ReST-MCTS*: LLM Self-Training via Process Reward Guided Tree Search","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.03816","snapshot_observed_at":"2026-08-11T12:41:20.737718Z","title":"https://openreview.net/forum?id=0NphYCmgua","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.737718Z"},"links":{"cited_paper":"/paper/2406.03816","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:d64c924efd45d93fc54c579f1a02813e4981db2992da3c50739ea0847a137943","observation_id":"da23cc91-d9c0-4101-9133-778167ea3036","resolution":{"observed_at":"2026-08-11T12:41:20.737718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-11T12:41:20.740628Z","title":"2311.07911","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.740628Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:0f0b7633c95cac66c7e240eb1619ebac0a22354234736844412c30ed6e3ea9f0","observation_id":"7194f92f-67ab-42f1-8ad8-6d1761891809","resolution":{"observed_at":"2026-08-11T12:41:20.740628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11411","last_updated":"2024-02-18T00:56:16Z","snapshot_observed_at":"2026-08-10T21:21:50.749962Z","submitted_at":"2024-02-18T00:56:16Z","title":"Aligning Modalities in Vision Large Language Models via Preference Fine-tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11411","snapshot_observed_at":"2026-08-11T12:41:20.743435Z","title":"https://doi.org/10.48550/arXiv.2402.11411","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.743435Z"},"links":{"cited_paper":"/paper/2402.11411","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:4b3283a10353051a02746ff298c08b9e1cdf1779d7ae29ba5f9880bcdb41a18d","observation_id":"bca6d087-992e-4ce6-af47-1f88ef4048e8","resolution":{"observed_at":"2026-08-11T12:41:20.743435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-11T12:41:20.715053Z","title":"Kaitao Song, Xu Tan, Tao Qin, Jianfeng Lu, and Tie-Yan Liu","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.715053Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:43d229d5fc0fb97f8324c95880190ec0bba7ea3d20d5ade242269a300eb07909","observation_id":"31b6813e-b368-44e8-b7c9-e9c86e13b50f","resolution":{"observed_at":"2026-08-11T12:41:20.715053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:41:20.916493Z","title":"Nisan Stiennon, Long Ouyang, Jeffrey Wu, Daniel M","venue":null,"work_id":"3998d72b-0287-446f-a2bf-4ac9cf681828","year":2020},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.717800Z"},"links":{"citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:307e88257a202a14c599b93ad26044cdffb101b4b655cb674cc5263a9ddd5b84","observation_id":"7a785059-adbd-411d-8285-ae033ae23db1","resolution":{"observed_at":"2026-08-11T12:41:20.920072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09279","last_updated":"2024-10-07T21:24:59Z","snapshot_observed_at":"2026-08-12T23:43:26.084884Z","submitted_at":"2024-06-13T16:17:21Z","title":"Unpacking DPO and PPO: Disentangling Best Practices for Learning from Preference Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09279","snapshot_observed_at":"2026-08-11T12:41:20.686480Z","title":"neurips.cc/paper/2021/hash/be83ab3ecd0db773eb2dc1b0a17836a1-Abstract-round2.html","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.686480Z"},"links":{"cited_paper":"/paper/2406.09279","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:012008a6be32a10731b373d784e2cbfb47ea647eb704061da0d10016e5f322af","observation_id":"d7dd16cd-5fd8-40b3-9895-2adc459dd902","resolution":{"observed_at":"2026-08-11T12:41:20.686480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-11T12:41:20.680428Z","title":"Ralph Bradley and Milton Terry","venue":null,"work_id":null,"year":1952},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.680428Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:59430b77ada2dac92706ea52d2d0eb7d374141f697eb0c641dba127b3e2a0f59","observation_id":"93cfca34-28bb-4a97-a4ac-f5d86d78c3f2","resolution":{"observed_at":"2026-08-11T12:41:20.680428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-11T12:41:20.683556Z","title":"2312.11805","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.683556Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:6ee539315fa56c1720fbd40e9a8762965c77a56527449dde26fec2af78e717fd","observation_id":"3c8ccd28-1f4c-40b4-aadd-3cd7f07c2584","resolution":{"observed_at":"2026-08-11T12:41:20.683556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-11T12:41:20.673423Z","title":"https: //doi.org/10.48550/arXiv.2407.21783","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.673423Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:63d804cf36aa4d83b7079ffef16c935b6561fc42be2e755a65b89d41859aad1d","observation_id":"0e0c2472-2aa9-484d-ae25-8e4a897602eb","resolution":{"observed_at":"2026-08-11T12:41:20.673423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-11T12:35:09.122376Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following"},"reference_resolution":{"displayed":26,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":24,"verified_exact":0,"verified_fuzzy":2},"total_outbound_references":26},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 26 of 26 outbound references and 0 inbound Pith citation observations for arXiv:2412.15282."}