{"as_of":"2026-08-21T05:58:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e19616a864e0fafcf9f819e62e9c5d9293245153a402b4ca1e8def842f7c2b9f","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T12:48:59.097529Z","state":"measured"},{"denominator":55,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":55,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T19:43:51.965882Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-10T22:35:49.296372Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"cited_work":{"arxiv_id":"2412.13862","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.13862","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Energy-based preference model offers better offline alignment than the bradley-terry preference model.arXiv preprint arXiv:2412.13862","venue":null,"work_id":"8550e340-937e-474e-9166-2ab68c87beae","year":null},"citing_paper":{"arxiv_id":"2604.04410","last_updated":"2026-04-06T04:21:24Z","snapshot_observed_at":"2026-08-13T00:39:01.752879Z","submitted_at":"2026-04-06T04:21:24Z","title":"Relative Density Ratio Optimization for Stable and Statistically Consistent Model Alignment","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T19:43:51.965882Z"},"links":{"cited_paper":"/paper/2412.13862","citing_paper":"/paper/2604.04410"},"observation_digest":"sha256:51e205dc617ff27ffc7dfddfe631b54726cb1356836c4e2223a235f049a9f873","observation_id":"7bc0c716-0e7d-4d43-9191-4dcd7c866d1e","resolution":{"observed_at":"2026-05-10T22:35:49.303293Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.13862/citation-record","integrity":"/paper/2412.13862/integrity","json":"/paper/2412.13862/citation-record.json","paper":"/paper/2412.13862"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.10571","last_updated":"2024-06-06T12:02:37Z","snapshot_observed_at":"2026-08-16T14:18:02.753710Z","submitted_at":"2024-02-16T10:55:38Z","title":"Direct Preference Optimization with an Offset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.10571","snapshot_observed_at":"2026-08-11T12:48:58.815716Z","title":"Direct preference optimization with an offset","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.815716Z"},"links":{"cited_paper":"/paper/2402.10571","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:ff3a2e1f8d3e5498e8672435e6142b1276455e4ac1495c23ed1593630fa29a6e","observation_id":"4e643310-f3c9-4ac2-9b82-7e2c1708c6fb","resolution":{"observed_at":"2026-08-11T12:48:58.815716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12036","last_updated":"2023-11-22T00:02:49Z","snapshot_observed_at":"2026-08-20T10:43:00.063759Z","submitted_at":"2023-10-18T15:21:28Z","title":"A General Theoretical Paradigm to Understand Learning from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12036","snapshot_observed_at":"2026-08-11T12:48:58.822647Z","title":"G., Rowland, M., Piot, B., Guo, D., Calandriello, D., Valko, M., and Munos, R","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.822647Z"},"links":{"cited_paper":"/paper/2310.12036","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:9cf5a45a6ea00b30b8ea77cd479896ff98d8c93b5e2a7bafa9b02dfb6458c2e2","observation_id":"15dea84b-555b-4a49-98f8-50a8e061c7ca","resolution":{"observed_at":"2026-08-11T12:48:58.822647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:49:00.049530Z","title":"and Rinaldo, A","venue":null,"work_id":"d79e0b6e-67ae-4b8f-8afe-6deac6ddabe7","year":2022},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.828616Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:c51eb5c3187cea3fdbf4f2b1f178bd0fd2921e2b97e6ad475ae3a79cb8225ac0","observation_id":"d45fb7b6-1d43-4ff6-93e1-9fbb14c543af","resolution":{"observed_at":"2026-08-11T12:49:00.055997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05369","last_updated":"2024-10-30T07:29:40Z","snapshot_observed_at":"2026-08-16T14:20:22.893368Z","submitted_at":"2024-02-08T02:58:47Z","title":"Noise Contrastive Alignment of Language Models with Explicit Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05369","snapshot_observed_at":"2026-08-11T12:48:58.834297Z","title":"Noise contrastive alignment of language models with explicit rewards","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.834297Z"},"links":{"cited_paper":"/paper/2402.05369","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:9a9f1a3f9bb0f58962e00ada46ee0fa36567744c768688741d227cb10f0d8835","observation_id":"ff5a3d7a-a53f-47e6-8719-9a5a608aa9e8","resolution":{"observed_at":"2026-08-11T12:48:58.834297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.01335","last_updated":"2024-06-14T21:17:17Z","snapshot_observed_at":"2026-08-19T08:48:56.371619Z","submitted_at":"2024-01-02T18:53:13Z","title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.01335","snapshot_observed_at":"2026-08-11T12:48:58.839933Z","title":"Self-play fine-tuning converts weak language models to strong language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.839933Z"},"links":{"cited_paper":"/paper/2401.01335","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:f4d72230a7f0428fd85963c55d44c7a7f91da34633ed1a2910ba23a87738162f","observation_id":"0e03fce6-5637-485c-bebf-5d07130da182","resolution":{"observed_at":"2026-08-11T12:48:58.839933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:58.846024Z","title":"F., Leike, J., Brown, T., Martic, M., Legg, S., and Amodei, D","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.846024Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:a9005660836f925c1efb8f58613f548c1fdebf8ab5a261cd68d2ea9219438094","observation_id":"4a858c7f-d256-4759-a084-cacf9b754644","resolution":{"observed_at":"2026-08-11T12:48:58.846024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-11T12:48:58.851831Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.851831Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:6b6248244cda65c243d78f67c6198a7e2faa79c6cf8ccd71030ac35a8f644eba","observation_id":"91b6e351-45dd-495b-bb75-784b1921c051","resolution":{"observed_at":"2026-08-11T12:48:58.851831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:58.857162Z","title":"Ultrafeedback: Boosting language models with scaled ai feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.857162Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:f7ec86d76b13d2200a0eaec36408d2a7d8348c3b8b744779803414e2224e09bc","observation_id":"1eaa6f73-9c6b-4db4-842a-de3ce7b7805d","resolution":{"observed_at":"2026-08-11T12:48:58.857162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:58.862036Z","title":"Learning discrete energy-based models via auxiliary-variable local exploration","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.862036Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:5f0a961bea895999469b5bee428d0977beb55d9f19d1bfaddd89d6dc240b405f","observation_id":"86a88aec-301e-405e-88f4-e4568f17e852","resolution":{"observed_at":"2026-08-11T12:48:58.862036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.992343Z","title":"Residual energy-based models for text generation","venue":null,"work_id":"a7c0df0d-6342-4635-a74f-3a1f6ca701f8","year":2020},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.866839Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:591d1bc0133055e95b48e57a4bafd824f8d4eba76a66dd18e275374bd143c34a","observation_id":"449f708d-6ae1-4dbf-adcb-6ce0e9346f12","resolution":{"observed_at":"2026-08-11T12:48:59.998228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.04475","last_updated":"2025-03-10T09:27:03Z","snapshot_observed_at":"2026-07-06T17:56:23.317089Z","submitted_at":"2024-04-06T02:29:02Z","title":"Length-Controlled AlpacaEval: A Simple Way to Debias Automatic Evaluators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.04475","snapshot_observed_at":"2026-08-11T12:48:58.872968Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.872968Z"},"links":{"cited_paper":"/paper/2404.04475","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:cec57a339c2dca70620ed13171960e7c52549f4e8c253c162234dc9b002648fd","observation_id":"9f44cb8c-944d-4a1a-9a66-29859631b523","resolution":{"observed_at":"2026-08-11T12:48:58.872968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.976379Z","title":"R., Elsahar, H., and Dymetman, M","venue":null,"work_id":"71fba18a-d5df-4a8a-a938-424a00879419","year":2022},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.878045Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:4b81c4fca837c1f92dc966ccf15cfaa7f771b1de8963fd874b7fb093dadaa9a1","observation_id":"d350913b-7986-4f91-888a-acf9c9761421","resolution":{"observed_at":"2026-08-11T12:48:59.981642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.959390Z","title":"Kto: Model alignment as prospect theoretic optimization, 2024","venue":null,"work_id":"8a97b724-73ff-4f28-8c0b-0450f549daf9","year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.882894Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:3a425d908a74251d108649f7afdaaba5c7c697febd9e5cc6dc2d55cbb722ea8d","observation_id":"828121dc-748d-4147-a966-1caf2d68abc2","resolution":{"observed_at":"2026-08-11T12:48:59.964578Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"stable/2308513","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.468242Z","title":null,"venue":null,"work_id":"7c14820a-7eb6-4c9b-b137-d8bfc20f87af","year":1957},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.887771Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:e7227431c77c4eabef7ac6a917b86c23b6ac422af258707a9e1e616db0308b0a","observation_id":"52dca5c1-f87d-4a03-ab07-7b7ca07815ca","resolution":{"observed_at":"2026-08-11T12:48:59.476360Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:58.892294Z","title":"Asymptotic theory of sparse Bradley–Terry model","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.892294Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:448780c4bbdba19a058690120872d3af3b0adee5e4d480b8de60015a363c2501","observation_id":"2c811697-94e4-4243-8405-b80341263d10","resolution":{"observed_at":"2026-08-11T12:48:58.892294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.939107Z","title":"Minimax rate for learning from pairwise comparisons in the BTL model","venue":null,"work_id":"3c6c4a7f-fa09-41c1-b7d5-ac0e3f419cd3","year":2020},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.897067Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:19ef0d2ae15a1ce1c0ad97c5c25c4d0864fb10d16433fe9636d04da05722d8e0","observation_id":"4cbb1edf-b085-4815-be1f-0829d05f24e3","resolution":{"observed_at":"2026-08-11T12:48:59.945388Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-11T12:48:58.901587Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.901587Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:e27f49b3b6d5b5f0ac1c01be30df0d193cfc9edf09b6cb3a86e04e96846a3f24","observation_id":"7eb9262a-763a-46d9-9b62-18ed5a70d129","resolution":{"observed_at":"2026-08-11T12:48:58.901587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07691","last_updated":"2024-03-14T07:47:08Z","snapshot_observed_at":"2026-08-16T14:56:35.674907Z","submitted_at":"2024-03-12T14:34:08Z","title":"ORPO: Monolithic Preference Optimization without Reference Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07691","snapshot_observed_at":"2026-08-11T12:48:58.906625Z","title":"Orpo: Monolithic preference optimization without reference model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.906625Z"},"links":{"cited_paper":"/paper/2403.07691","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:0452dfae78ccc84539ebc587e18092b287654ba988c55c81f7f3b5eeced9d2df","observation_id":"53924ce4-02b5-4848-a621-1a27c7fda0d4","resolution":{"observed_at":"2026-08-11T12:48:58.906625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.921615Z","title":"Some extensions of score matching","venue":null,"work_id":"83943a65-18f2-44f6-be8a-8c08db374858","year":2007},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.911994Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:38ef0c76f62f89278eb13ba775191f3d4a732f70813be665f36fc9c100edd0e3","observation_id":"116d37b5-abc7-4032-8e85-b3459118f259","resolution":{"observed_at":"2026-08-11T12:48:59.927093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.04656","last_updated":"2025-06-09T07:10:33Z","snapshot_observed_at":"2026-08-19T04:57:34.223598Z","submitted_at":"2024-04-06T15:20:59Z","title":"Binary Classifier Optimization for Large Language Model Alignment","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.04656","snapshot_observed_at":"2026-08-11T12:48:58.916634Z","title":"W., and On, K.-W","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.916634Z"},"links":{"cited_paper":"/paper/2404.04656","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:9057305e5ed2ec022cf15347f974b060d48bed6433f9cb9829cbd64d9c2faf9a","observation_id":"9e6571af-398a-4cd2-9ca0-bbb2bd6782d2","resolution":{"observed_at":"2026-08-11T12:48:58.916634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:58.921656Z","title":"and Langford, J","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.921656Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:575770a30161669f7d5c4ed638e3e34bbefbd70050f01460ec2bdf6d9a922d48","observation_id":"e0a39171-716c-4599-8fc1-3f1779848bcf","resolution":{"observed_at":"2026-08-11T12:48:58.921656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.893149Z","title":"A distributional approach to controlled text generation","venue":null,"work_id":"394a201e-121e-41aa-beff-4563546b72b1","year":2021},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.926599Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:723f45b4594af4379f836a800b0f0531a9ef15f5ab2586ce159cbc5b3810240a","observation_id":"2bc77cfa-62e5-49d4-829c-a4b05839c4b9","resolution":{"observed_at":"2026-08-11T12:48:59.898378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.875016Z","title":"On reinforcement learning and distribution matching for fine-tuning language models with no catastrophic forgetting","venue":null,"work_id":"1e7f8c39-87d3-44d2-94ab-60bf1f04936d","year":2022},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.931285Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:73ee467016f4df3184dea040406243fda962c48c96afd0749370707e46dbc3e2","observation_id":"b94f1ae7-803b-44f3-b269-aba4a3ce4182","resolution":{"observed_at":"2026-08-11T12:48:59.881429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.857553Z","title":"Rl with kl penalties is better viewed as bayesian inference","venue":null,"work_id":"100d104c-3e36-475f-a846-dd8a4fe94baa","year":2022},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.936142Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:eaffe13867a28353dd3b44d7e947445ba70934cccacc191eafd1272f303eeca3","observation_id":"58601b35-69b5-46ac-afd6-63283f1c592c","resolution":{"observed_at":"2026-08-11T12:48:59.863106Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:58.941058Z","title":"Crafting papers on machine learning","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.941058Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:7480807dcae62fa8149897e3ec2f0e906a8618984ffc65028b9d2dee5db5c5ef","observation_id":"2dbdaf73-3119-4d2a-bded-a33d6ce348aa","resolution":{"observed_at":"2026-08-11T12:48:58.941058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.830392Z","title":"Perturb-and-max-product: Sampling and learning in discrete energy-based models","venue":null,"work_id":"0360d0b3-a524-44c9-9d9a-1edfea90e3d0","year":2021},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.946462Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:23eaecdb73766c6f35e9e64e2ad7e6e7941905fcbf354f6f3ad3ba116b27731c","observation_id":"3df27b0e-663b-4639-9ee5-cc8616648eff","resolution":{"observed_at":"2026-08-11T12:48:59.835544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:58.951718Z","title":"A tutorial on energy-based learning","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.951718Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:bd75cc18eca160a7dc5bbe8062a46353998fc4c59355348c2c8b1ddda9b7aef6","observation_id":"ab292aad-ea84-47e1-a8ed-acbfbb8fae1e","resolution":{"observed_at":"2026-08-11T12:48:58.951718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.802349Z","title":"Conditional strong law of large number","venue":null,"work_id":"1b093e22-b0ac-4c5f-b7b3-b13a7bb3256e","year":2005},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.956555Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:4fd620d242bb8efbfc5de99127d932203ff2913dc8085c9dbcf202a09d1ba614","observation_id":"08f6e237-f9ba-4938-a3d4-ba5b03f99b89","resolution":{"observed_at":"2026-08-11T12:48:59.807814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.786403Z","title":"Concrete score matching: Generalized score matching for discrete data","venue":null,"work_id":"cb924289-8007-454e-a5f9-36879656fc9c","year":2022},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.961658Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:4e678784a77df1d4908386210a343bb5ae2eeb7e8090d1f3dce02633c6e2f1ba","observation_id":"e7ad1929-501a-4d4d-ba23-7a817d6d9d1f","resolution":{"observed_at":"2026-08-11T12:48:59.791576Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14734","last_updated":"2024-11-01T20:05:19Z","snapshot_observed_at":"2026-08-20T05:24:55.574908Z","submitted_at":"2024-05-23T16:01:46Z","title":"SimPO: Simple Preference Optimization with a Reference-Free Reward","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14734","snapshot_observed_at":"2026-08-11T12:48:58.966786Z","title":"Simpo: Simple preference optimization with a reference-free reward","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.966786Z"},"links":{"cited_paper":"/paper/2405.14734","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:8458b70d51a70b9c9f5fc9da2a019f07c853b4c6188572b383ec88df7066e78f","observation_id":"5b26f855-819d-4343-aa5e-9c62415b335f","resolution":{"observed_at":"2026-08-11T12:48:58.966786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.768890Z","title":"A note on dpo with noisy preferences & relationship to ipo, 2023","venue":null,"work_id":"9bda69ab-9367-4622-a3f6-bb4a1a3eb542","year":2023},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.972468Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:a5e800ab0c6aa1840bbec828f54425e879cba8ed2bc858c8279bb01c270cc865","observation_id":"4ad4fb31-89a4-46d4-b5d0-9b9ec9d784c6","resolution":{"observed_at":"2026-08-11T12:48:59.774846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:58.978098Z","title":"and Szepesv \\'a ri, C","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.978098Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:77f93753c00ecca6615a05c6568cf42a1a83db89ae3420a8d2d0545d93ade4e5","observation_id":"15159972-dbd5-4056-8625-2a0397895fa9","resolution":{"observed_at":"2026-08-11T12:48:58.978098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:58.983342Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.983342Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:5c5f65f6525ba02e8e03892245e470e8b3c442bbed65b5d3945f13dad343bdd8","observation_id":"005ebca7-3c69-4b12-b83b-9af5fd11e260","resolution":{"observed_at":"2026-08-11T12:48:58.983342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19159","last_updated":"2024-09-09T04:39:31Z","snapshot_observed_at":"2026-08-20T11:07:35.159951Z","submitted_at":"2024-03-28T06:03:47Z","title":"Disentangling Length from Quality in Direct Preference Optimization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19159","snapshot_observed_at":"2026-08-11T12:48:58.989023Z","title":"Disentangling length from quality in direct preference optimization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.989023Z"},"links":{"cited_paper":"/paper/2403.19159","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:afeb5759cdd113aced1cefad3f57374c7beba35fe91e62d906ef8e1a5c2318f0","observation_id":"8f291d2b-6762-4ace-80c8-727e3371753d","resolution":{"observed_at":"2026-08-11T12:48:58.989023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.731605Z","title":"Distributional reinforcement learning for energy-based sequential models","venue":null,"work_id":"dc488fbf-985c-4f8c-acc7-50c1d6dd5220","year":2019},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.994572Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:01c0a5f2fdcbedfa228782ac6ffbe218db655567cb12ddb22bc83e98788c3ad4","observation_id":"dce41a7c-42ea-430f-b70b-1f3ecee24051","resolution":{"observed_at":"2026-08-11T12:48:59.736327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.714884Z","title":"Red teaming language models with language models","venue":null,"work_id":"ce2a7160-87bf-4c9d-bace-0b344030b7dd","year":2022},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:58.999638Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:5739af1597a798e38e2062b06c7167a4632f1675d04ea566825b93e831683fa0","observation_id":"b82470cd-8163-41d7-991a-4e2e314da0a4","resolution":{"observed_at":"2026-08-11T12:48:59.720009Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.006303Z","title":"D., Ermon, S., and Finn, C","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.006303Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:80119b732d3489d6ab5e25e8769aa9b58a1e0dd742081ee76d681908893baad0","observation_id":"dcd78151-8bd9-4e8c-a7d4-3ac44e66082a","resolution":{"observed_at":"2026-08-11T12:48:59.006303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.689086Z","title":null,"venue":null,"work_id":"ce7037c9-d268-4645-924e-9b624a5680c7","year":2023},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.011961Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:7211ce46bf75176db46688fd830c6e6ec5ea4cc51b15a2ba373e27fcb337b1c2","observation_id":"ffa53833-34fc-4044-b7bc-e5ddbf7c1d55","resolution":{"observed_at":"2026-08-11T12:48:59.694059Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.673231Z","title":"and Yao, Y.-C","venue":null,"work_id":"b0fc4a18-2563-4100-83f9-22ae00f91d6e","year":1999},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.017603Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:5ef48b5ef534503675e5a2a81a6368c4d92350f76565bee41949d1ccfae68c00","observation_id":"037f79df-dffe-4495-b58c-a2ae781b06ed","resolution":{"observed_at":"2026-08-11T12:48:59.678282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.657306Z","title":"Preference ranking optimization for human alignment","venue":null,"work_id":"26469257-fdb1-447a-8fd9-28b2c814a3fd","year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.023076Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:af84f8a04718bf7471cdaf73c6b5961168d8bef0e6eae59675fdd650ef956de3","observation_id":"b38a4a33-1473-433a-ac39-ebf1467aa183","resolution":{"observed_at":"2026-08-11T12:48:59.662482Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2101.03288","last_updated":"2021-02-17T19:20:09Z","snapshot_observed_at":"2026-08-16T18:53:23.567847Z","submitted_at":"2021-01-09T04:51:31Z","title":"How to Train Your Energy-Based Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.03288","snapshot_observed_at":"2026-08-11T12:48:59.028265Z","title":"and Kingma, D","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.028265Z"},"links":{"cited_paper":"/paper/2101.03288","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:4605e00e19b5c84b3b0371e32836f3250ffe42aef0fe2d99eb48b6de67e3d311","observation_id":"fdc8a931-4580-4b3f-954a-bbd70659ff33","resolution":{"observed_at":"2026-08-11T12:48:59.028265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.034333Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.034333Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:3f2346996771b9c0a7e908d1798ec12b7e5859bdbf4dbc868d5e28487d2427cc","observation_id":"fd091cf0-8cdd-4fed-8e57-d72ee52be94b","resolution":{"observed_at":"2026-08-11T12:48:59.034333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05749","last_updated":"2024-05-28T23:25:15Z","snapshot_observed_at":"2026-08-16T14:20:11.482387Z","submitted_at":"2024-02-08T15:33:09Z","title":"Generalized Preference Optimization: A Unified Approach to Offline Alignment","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05749","snapshot_observed_at":"2026-08-11T12:48:59.040041Z","title":"D., Zheng, Z., Calandriello, D., Munos, R., Rowland, M., Richemond, P","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.040041Z"},"links":{"cited_paper":"/paper/2402.05749","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:b37258908fca41bdc303627a2dd1b111015a2967e105b4429188c44e840b119a","observation_id":"77207307-2b79-4c21-928d-6557bafce82e","resolution":{"observed_at":"2026-08-11T12:48:59.040041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.12066","last_updated":"2021-11-30T22:47:59Z","snapshot_observed_at":"2026-08-18T22:21:48.609989Z","submitted_at":"2021-06-22T21:25:43Z","title":"It's All in the Heads: Using Attention Heads as a Baseline for Cross-Lingual Transfer in Commonsense Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.12066","snapshot_observed_at":"2026-08-11T12:48:59.045427Z","title":"and Ryabinin, M","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.045427Z"},"links":{"cited_paper":"/paper/2106.12066","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:b369db0a32b23d408175ce94d76ee069ce3d3755e1aae817232b884c1f317526","observation_id":"ee60719a-a4d6-430a-af2a-150cba8896c4","resolution":{"observed_at":"2026-08-11T12:48:59.045427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16944","last_updated":"2023-10-25T19:25:16Z","snapshot_observed_at":"2026-08-18T22:05:54.705341Z","submitted_at":"2023-10-25T19:25:16Z","title":"Zephyr: Direct Distillation of LM Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16944","snapshot_observed_at":"2026-08-11T12:48:59.050458Z","title":"Zephyr: Direct distillation of lm alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.050458Z"},"links":{"cited_paper":"/paper/2310.16944","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:06828e73800c015f8ddaa0d59c4a761438dfb80f9bdbff5e6602bbf5b307dac8","observation_id":"4e4f9b66-1c89-4f85-b756-cda3b00f5b2f","resolution":{"observed_at":"2026-08-11T12:48:59.050458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.04341","last_updated":"2022-05-09T14:41:35Z","snapshot_observed_at":"2026-08-16T17:01:36.416348Z","submitted_at":"2022-05-09T14:41:35Z","title":"Asymptotic comparison of identifying constraints for Bradley-Terry models","version":1},"cited_work":{"arxiv_id":"2205.04341","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.04341","snapshot_observed_at":"2026-08-11T12:48:59.198781Z","title":"Asymptotic comparison of identifying constraints for Bradley-Terry models","venue":"math.ST","work_id":"5670252c-42d5-4cf5-bb23-6d8f6f20baf9","year":2022},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.055874Z"},"links":{"cited_paper":"/paper/2205.04341","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:ac0791109075bb19004fdfe2d20cb627cd4e71eb1f780365b23d6bd0d01d6548","observation_id":"01e071f4-380c-4e77-bf70-8461579cce64","resolution":{"observed_at":"2026-08-11T12:48:59.205867Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.08417","last_updated":"2024-06-03T01:28:06Z","snapshot_observed_at":"2026-08-20T10:44:37.347378Z","submitted_at":"2024-01-16T15:04:51Z","title":"Contrastive Preference Optimization: Pushing the Boundaries of LLM Performance in Machine Translation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.08417","snapshot_observed_at":"2026-08-11T12:48:59.061553Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.061553Z"},"links":{"cited_paper":"/paper/2401.08417","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:a206a463607e85568452fb01557a6bf8295c92b415da72f3028fc68a96383f33","observation_id":"fe111916-a9b7-41d7-aa9b-a64bb9ef0bf7","resolution":{"observed_at":"2026-08-11T12:48:59.061553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.630784Z","title":"Q., Salamatian, S., Sun, Z., Suresh, A","venue":null,"work_id":"8a98f56e-7177-4a16-8666-8a17949a4e16","year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.067084Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:c5e366a5a7afc4c26f75950aa45c1a523513e73d203e7dfe4482e24a17f5fb5f","observation_id":"fbe02fec-a988-4395-8111-dad8e2699249","resolution":{"observed_at":"2026-08-11T12:48:59.635861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.071970Z","title":"RRHF : Rank responses to align language models with human feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.071970Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:064cf26a3795e4ab4de0d36050face8ea81ece3fba25a4f72c15efd022d1fc93","observation_id":"5d502206-73bc-4899-93c3-2aa4911776e8","resolution":{"observed_at":"2026-08-11T12:48:59.071970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.077015Z","title":"Offline reinforcement learning with realizability and single-policy concentrability","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.077015Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:cfe76f73a760d9f8cd338f7460a332a8234b5469b89bd4bb06e644dc731d8393","observation_id":"0c8bd438-bdf2-4a90-b12a-3584b9180ecf","resolution":{"observed_at":"2026-08-11T12:48:59.077015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.082195Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.082195Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:69d0485578d4bf23a03404ef1359840bafba1fcdbc780b12165cd67ac91c9910","observation_id":"6a98a1c6-2865-4970-b9b2-d61bd2128927","resolution":{"observed_at":"2026-08-11T12:48:59.082195Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11827","last_updated":"2024-10-03T21:37:02Z","snapshot_observed_at":"2026-08-16T13:42:07.691322Z","submitted_at":"2024-06-17T17:59:13Z","title":"WPO: Enhancing RLHF with Weighted Preference Optimization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11827","snapshot_observed_at":"2026-08-11T12:48:59.087392Z","title":"R., Zhao, S., Song, K., Xu, S., and Zhu, C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.087392Z"},"links":{"cited_paper":"/paper/2406.11827","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:c3bb8d1239100a0d42482b71cd0714e3a0f529a2be166fc3d372862fb285ae7c","observation_id":"5051a299-b8c4-4b80-b238-4c089ae2725a","resolution":{"observed_at":"2026-08-11T12:48:59.087392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08593","last_updated":"2020-01-08T23:02:36Z","snapshot_observed_at":"2026-08-16T00:15:57.597094Z","submitted_at":"2019-09-18T17:33:39Z","title":"Fine-Tuning Language Models from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08593","snapshot_observed_at":"2026-08-11T12:48:59.092277Z","title":"M., Stiennon, N., Wu, J., Brown, T","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.092277Z"},"links":{"cited_paper":"/paper/1909.08593","citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:0947c94a33e857559586441722655ea64f55e11d66bfbbddbe89a58bbcf89b3b","observation_id":"9e76b640-4a81-4fd8-b7cd-d66c14432c06","resolution":{"observed_at":"2026-08-11T12:48:59.092277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:48:59.097529Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-11T12:48:59.097529Z"},"links":{"citing_paper":"/paper/2412.13862"},"observation_digest":"sha256:3168c48664133222446e43148897892d4a3b9a38a3682a653ec847aed9aef2cd","observation_id":"0d4675f9-bdff-4f27-a786-747aa81bd8a7","resolution":{"observed_at":"2026-08-11T12:48:59.097529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.13862","last_updated":"2024-12-18T13:55:42Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-14T07:37:10.037547Z","submitted_at":"2024-12-18T13:55:42Z","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":34,"verified_exact":2,"verified_fuzzy":18},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 1 inbound Pith citation observation for arXiv:2412.13862."}