{"as_of":"2026-08-08T23:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:59aeedc7087c17a3fe79542c7a67d5c3bf14912962992458a131ace694800f4f","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":13,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":13,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":13,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":13,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T11:20:48.489366Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-21T07:59:50.977785Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-08T11:20:48.489366Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.07956","last_updated":"2025-02-11T21:09:24Z","snapshot_observed_at":"2026-08-08T11:16:02.213927Z","submitted_at":"2025-02-11T21:09:24Z","title":"Bridging HCI and AI Research for the Evaluation of Conversational SE Assistants","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-08T11:20:48.489366Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2502.07956"},"observation_digest":"sha256:a9aa692adbf0f2ed6256097e573cf0bba715cd67d12717a3fd254575c36cda57","observation_id":"b5c40558-960e-416e-b402-753a034c8944","resolution":{"observed_at":"2026-08-08T11:20:48.489366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-07T14:31:47.967756Z","title":"Arithmetic control of llms for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21537","last_updated":"2025-05-24T09:07:13Z","snapshot_observed_at":"2026-08-08T02:37:32.325481Z","submitted_at":"2025-05-24T09:07:13Z","title":"OpenReview Should be Protected and Leveraged as a Community Asset for Research in the Era of Large Language Models","version":1},"reference_index":151,"source":"arxiv_source","source_observed_at":"2026-08-07T14:31:47.967756Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2505.21537"},"observation_digest":"sha256:a90beed6c40782a07b565b5762a58467081bf68e784fcd9b92f538803368e653","observation_id":"2c8814b4-ac09-42a0-8df4-2e87ae403b07","resolution":{"observed_at":"2026-08-07T14:31:47.967756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-06T18:10:18.299705Z","title":"Arithmetic control of llms for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09060","last_updated":"2025-07-15T17:48:41Z","snapshot_observed_at":"2026-08-08T09:58:17.034055Z","submitted_at":"2025-07-11T22:33:11Z","title":"CALMA: A Process for Deriving Context-aligned Axes for Language Model Alignment","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T18:10:18.299705Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2507.09060"},"observation_digest":"sha256:432b9cf7156e41804c7f5c3b5807113680d4fd92e767aed1a984bdee3668f3c5","observation_id":"e7c8015e-3c13-4bf1-bfbc-7df13e0bfdea","resolution":{"observed_at":"2026-08-06T18:10:18.299705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2507.17746","last_updated":"2025-10-03T01:55:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-23T17:57:55Z","title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T06:07:56.678339Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2507.17746"},"observation_digest":"sha256:7493f471dbb52c651536272ce7c80b4f5d6c7084cedb09ee4e1b676845d2c8d9","observation_id":"455ce814-39ff-4b23-ac00-f1f8cb8a6233","resolution":{"observed_at":"2026-05-13T06:07:56.859644Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2604.24536","last_updated":"2026-04-27T14:33:45Z","snapshot_observed_at":"2026-07-06T23:10:33.821313Z","submitted_at":"2026-04-27T14:33:45Z","title":"Generating Place-Based Compromises Between Two Points of View","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-05-08T03:36:31.695964Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2604.24536"},"observation_digest":"sha256:8ab2cae8448f3f44efa972b011eeabd41159a059a7e4ee9d4eaa66136fbc5371","observation_id":"6f690dbf-7c73-41e2-8399-c55da0741e86","resolution":{"observed_at":"2026-05-11T22:01:12.114202Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.06987","last_updated":"2026-05-07T22:05:23Z","snapshot_observed_at":"2026-07-31T05:30:22.249144Z","submitted_at":"2026-05-07T22:05:23Z","title":"Response Time Enhances Alignment with Heterogeneous Preferences","version":1},"reference_index":164,"source":"arxiv_source","source_observed_at":"2026-05-11T01:04:26.288913Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.06987"},"observation_digest":"sha256:b4b6809327a72cff6588ffc3e09b18156ef545532f8cc2c85defd34e486dfb31","observation_id":"b3524346-517f-4318-8abe-cf6999948d15","resolution":{"observed_at":"2026-05-11T04:45:59.808702Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.07162","last_updated":"2026-05-08T02:47:30Z","snapshot_observed_at":"2026-07-06T23:19:35.885430Z","submitted_at":"2026-05-08T02:47:30Z","title":"CLIPer: Tailoring Diverse User Preference via Classifier-Guided Inference-Time Personalization","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-11T02:16:28.593349Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.07162"},"observation_digest":"sha256:5f7072249160146d55eca57e42d6d8ec85208dfdee3ab9d083c069fe93fe43a0","observation_id":"e1eb872b-f9b3-4eca-b37c-7cdbd9eec21d","resolution":{"observed_at":"2026-05-11T03:50:54.299657Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.11679","last_updated":"2026-05-13T09:28:34Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T07:38:59Z","title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-13T01:03:10.263663Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.11679"},"observation_digest":"sha256:7fbc5ec1f42209acfb82c6a2992a8d9dcc645c5fbccd9b59db799ffc57ebd985","observation_id":"f5a28b24-336f-4925-9cc1-7de381f28271","resolution":{"observed_at":"2026-05-13T01:57:06.526224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.11679","last_updated":"2026-05-13T09:28:34Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T07:38:59Z","title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-14T21:12:06.989077Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.11679"},"observation_digest":"sha256:11acca2488fc6082e0b7e0c029931d9df465353b5df40950732e617b231296e0","observation_id":"7675384e-1375-4dae-9231-c096f722abec","resolution":{"observed_at":"2026-05-14T21:12:58.887870Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.19330","last_updated":"2026-05-19T04:07:41Z","snapshot_observed_at":"2026-07-06T23:30:06.876413Z","submitted_at":"2026-05-19T04:07:41Z","title":"MOCHA: Multi-Objective Chebyshev Annealing for Agent Skill Optimization","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-20T06:09:56.684622Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.19330"},"observation_digest":"sha256:208f68831cd48c6c201445cf260a532c912f4d459aa2ce7519f60fded31a0b4d","observation_id":"1de03b3c-9dc8-4e29-bbd7-4c767030b6f2","resolution":{"observed_at":"2026-05-20T06:13:05.313131Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":"2402.18571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Arithmetic control of LLMs for diverse user preferences: Directional preference alignment with multi-objective rewards","venue":null,"work_id":"891ab6ad-8181-4659-9ba7-8096330036f5","year":2024},"citing_paper":{"arxiv_id":"2605.20408","last_updated":"2026-05-19T19:04:47Z","snapshot_observed_at":"2026-07-06T23:30:58.549353Z","submitted_at":"2026-05-19T19:04:47Z","title":"Spectral Souping: A Unified Framework for Online Preference Alignment","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-21T07:54:56.356555Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2605.20408"},"observation_digest":"sha256:8fbd12856ec0549c6381327bd9e464dd7a665f11ba852910d9d941d3bba6c992","observation_id":"bb1526a0-e8e7-49a9-8bfd-b38fd5a1973e","resolution":{"observed_at":"2026-05-21T07:59:50.979793Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-07-11T13:53:36.775836Z","title":"arXiv preprint arXiv:2402.18571 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-02T10:24:43.977557Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":1},"reference_index":148,"source":"arxiv_source","source_observed_at":"2026-07-11T13:53:36.775836Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:52b50f6994672863abcf347254090a08d117a18affe83d1bda387cb4f0d95fcc","observation_id":"dfaca053-4c67-49e6-b91a-07cf2c2eab44","resolution":{"observed_at":"2026-07-11T13:53:36.775836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18571","snapshot_observed_at":"2026-08-02T08:40:48.709771Z","title":"arXiv preprint arXiv:2402.18571 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-02T10:24:43.977557Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":3},"reference_index":149,"source":"arxiv_source","source_observed_at":"2026-08-02T08:40:48.709771Z"},"links":{"cited_paper":"/paper/2402.18571","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:0a799e606f6bd27df19e3673d248e8b4350afb0538eddc48bc8ea0abaecea6de","observation_id":"364a638b-056d-4e49-bf1e-b4ceb9c665a2","resolution":{"observed_at":"2026-08-02T08:40:48.709771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2402.18571/citation-record","integrity":"/paper/2402.18571/integrity","json":"/paper/2402.18571/citation-record.json","paper":"/paper/2402.18571"},"outbound":[],"paper":{"arxiv_id":"2402.18571","last_updated":"2024-03-06T08:07:02Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T17:37:01.208330Z","submitted_at":"2024-02-28T18:58:25Z","title":"Arithmetic Control of LLMs for Diverse User Preferences: Directional Preference Alignment with Multi-Objective Rewards"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 13 inbound Pith citation observations for arXiv:2402.18571."}