{"as_of":"2026-08-10T09:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0ec376de4b8744dfa95643913db7cdfebef344e35c4ccddae20a9df63157ddeb","coverage":[{"denominator":17,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:14:16.727994Z","state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.21131/citation-record","integrity":"/paper/2507.21131/integrity","json":"/paper/2507.21131/citation-record.json","paper":"/paper/2507.21131"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1606.06565","last_updated":"2016-07-25T17:23:29Z","snapshot_observed_at":"2026-07-06T05:00:46.434335Z","submitted_at":"2016-06-21T13:37:05Z","title":"Concrete Problems in AI Safety","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.06565","snapshot_observed_at":"2026-08-06T15:14:16.642328Z","title":"Concrete problems in ai safety","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.642328Z"},"links":{"cited_paper":"/paper/1606.06565","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:3f2b2ef00b58c57d87fc3ba8a49cffc147fc55b5048e44ca13b679242fa00fc0","observation_id":"98fea892-9983-4f3c-bf58-55ae314d04c5","resolution":{"observed_at":"2026-08-06T15:14:16.642328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1805.00899","last_updated":"2018-10-22T17:36:07Z","snapshot_observed_at":"2026-08-02T15:33:17.783178Z","submitted_at":"2018-05-02T16:27:32Z","title":"AI safety via debate","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1805.00899","snapshot_observed_at":"2026-08-06T15:14:16.663769Z","title":"Ai safety via debate.arXiv preprint arXiv:1805.00899,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.663769Z"},"links":{"cited_paper":"/paper/1805.00899","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:f2ac2e1c082ceb16ca8e6c57336e20ea7d56705b8a3a47d65877ab8d118aa51e","observation_id":"e5adc283-597b-4b00-82c4-e31932f82b47","resolution":{"observed_at":"2026-08-06T15:14:16.663769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.05221","last_updated":"2022-11-21T16:38:35Z","snapshot_observed_at":"2026-08-06T08:34:11.887259Z","submitted_at":"2022-07-11T22:59:39Z","title":"Language Models (Mostly) Know What They Know","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.05221","snapshot_observed_at":"2026-08-06T15:14:16.669363Z","title":"Language models struggle to generalize alignment from training","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.669363Z"},"links":{"cited_paper":"/paper/2207.05221","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:81cac341713bc5a13002e2f714d5fc99fa88fce6215914a32fa604420e1452fe","observation_id":"211cf569-56ac-4ca5-b6fb-5609d1d9d85e","resolution":{"observed_at":"2026-08-06T15:14:16.669363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:14:17.066532Z","title":"Algorithms for inverse reinforcement learn- ing","venue":null,"work_id":"a0f1304b-df4d-482a-85ec-32c2d975fc2b","year":2000},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.684460Z"},"links":{"citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:bcbe4fb4551d5830e415d24139a5c73b58740c4829323e3e1fdf474eac6767e0","observation_id":"19245a6b-857e-4a22-b0fa-cae434bd318c","resolution":{"observed_at":"2026-08-06T15:14:17.071285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18290","last_updated":"2024-07-29T22:26:36Z","snapshot_observed_at":"2026-08-01T16:34:38.795326Z","submitted_at":"2023-05-29T17:57:46Z","title":"Direct Preference Optimization: Your Language Model is Secretly a Reward Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.18290","snapshot_observed_at":"2026-08-06T15:14:16.693748Z","title":"Direct preference opti- mization: Your language model is secretly a reward model.arXiv preprint arXiv:2305.18290,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.693748Z"},"links":{"cited_paper":"/paper/2305.18290","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:06c327e12596f4606319230c54ec5ea446cd578f3b678ef9c634a587686b3d32","observation_id":"436d7b01-70a3-4706-abef-90a5af555369","resolution":{"observed_at":"2026-08-06T15:14:16.693748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1812.03030","last_updated":"2019-08-27T04:28:47Z","snapshot_observed_at":"2026-08-06T21:08:47.855179Z","submitted_at":"2018-11-30T08:28:38Z","title":"A new system-wide diversity measure for recommendations with efficient algorithms","version":2},"cited_work":{"arxiv_id":"1812.03030","doi":null,"metadata_source":"pith","pith_arxiv_id":"1812.03030","snapshot_observed_at":"2026-08-06T15:14:16.823705Z","title":"A new system-wide diversity measure for recommendations with efficient algorithms","venue":"cs.IR","work_id":"f8bc13d6-80f7-420e-9bf9-f0f67fed91ae","year":2018},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.703719Z"},"links":{"cited_paper":"/paper/1812.03030","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:06e2d6a888b4408ff5ad22caead43f044a6a1990875efbcacc4892afb102b663","observation_id":"06b5a3b5-ecfd-448d-91dd-f902350515fc","resolution":{"observed_at":"2026-08-06T15:14:16.828662Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.06083","last_updated":"2020-08-19T18:07:40Z","snapshot_observed_at":"2026-08-06T21:55:39.879833Z","submitted_at":"2020-04-13T17:32:05Z","title":"The Fates of Merging Supermassive Black Holes and a Proposal for a New Class of X-Ray Sources","version":3},"cited_work":{"arxiv_id":"2004.06083","doi":null,"metadata_source":"pith","pith_arxiv_id":"2004.06083","snapshot_observed_at":"2026-08-06T15:14:16.795897Z","title":"The Fates of Merging Supermassive Black Holes and a Proposal for a New Class of X-Ray Sources","venue":"astro-ph.GA","work_id":"42ddaad8-0ba6-4b52-810c-0169cf415c9d","year":2020},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.708646Z"},"links":{"cited_paper":"/paper/2004.06083","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:8161622caaeef30cfbc6f6aea51f9140c6d715e43b0dfacf7e21f1019514f9d8","observation_id":"0fe0c28a-d246-4698-861e-37cba15abb66","resolution":{"observed_at":"2026-08-06T15:14:16.803484Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:14:17.028977Z","title":"Red Button","venue":null,"work_id":"09fc904c-09de-4f0c-89e5-14b17d4d2ba3","year":2016},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.727994Z"},"links":{"citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:8b75f5f83f93f3908fe9f42628b78cbcdde45a271bac0ada17a8cf32da9e2376","observation_id":"02197dd9-d428-456a-b78c-87ad2378f729","resolution":{"observed_at":"2026-08-06T15:14:17.034115Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.03827","last_updated":"2024-03-02T21:33:53Z","snapshot_observed_at":"2026-08-03T01:53:33.505931Z","submitted_at":"2022-12-07T18:17:56Z","title":"Discovering Latent Knowledge in Language Models Without Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.03827","snapshot_observed_at":"2026-08-06T15:14:16.689102Z","title":"Discovering latent knowledge in language models without supervision","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2000,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.689102Z"},"links":{"cited_paper":"/paper/2212.03827","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:67ee70f801a1de30301267022e04eeca10fc71966b51149985fab4bc3b97aa27","observation_id":"8a1a899e-d56f-4157-9daa-0fef7608adf1","resolution":{"observed_at":"2026-08-06T15:14:16.689102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-06T15:14:16.648067Z","title":"Training a helpful and harmless assistant with rlhf","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.648067Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:2a1a15413a3473f2e25744f8bb062e780fdec9fb99efc2d3fd8cf166a3f719df","observation_id":"dd39c946-5aac-4e79-aa74-8da336e6e031","resolution":{"observed_at":"2026-08-06T15:14:16.648067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.08575","last_updated":"2018-10-19T16:30:48Z","snapshot_observed_at":"2026-08-04T21:33:05.702276Z","submitted_at":"2018-10-19T16:30:48Z","title":"Supervising strong learners by amplifying weak experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.08575","snapshot_observed_at":"2026-08-06T15:14:16.652974Z","title":"Supervising strong learners by amplifying weak experts","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.652974Z"},"links":{"cited_paper":"/paper/1810.08575","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:6c295c3ae548b7681864199193ff8188470323df383bff3ac45c77068ac18d58","observation_id":"ad369e9c-2031-4c53-8021-f4544014d216","resolution":{"observed_at":"2026-08-06T15:14:16.652974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.14375","last_updated":"2022-09-28T19:04:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-28T19:04:43Z","title":"Improving alignment of dialogue agents via targeted human judgements","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.14375","snapshot_observed_at":"2026-08-06T15:14:16.657949Z","title":"Improving align- ment of dialogue agents via targeted human judgements","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.657949Z"},"links":{"cited_paper":"/paper/2209.14375","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:4ada9b4d9a1814aeb5bc5469b581547eca4987e70cd7d41da8ad69a965294683","observation_id":"a1494bd8-4baf-4426-8e33-f0d8c6450a55","resolution":{"observed_at":"2026-08-06T15:14:16.657949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.05802","last_updated":"2022-06-14T01:16:24Z","snapshot_observed_at":"2026-07-06T13:19:58.934755Z","submitted_at":"2022-06-12T17:40:53Z","title":"Self-critiquing models for assisting human evaluators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.05802","snapshot_observed_at":"2026-08-06T15:14:16.698792Z","title":"Self-critiquing models for assisting human evaluators.arXiv preprint arXiv:2206.05802,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.698792Z"},"links":{"cited_paper":"/paper/2206.05802","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:77c5a5ad70b532686f9df1d27bd212695bd419b34ad355e8628a5643f15b9aa7","observation_id":"4cbba20d-fc0e-45f2-833b-801d7b7fbe1b","resolution":{"observed_at":"2026-08-06T15:14:16.698792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02231","last_updated":"2023-10-03T17:41:46Z","snapshot_observed_at":"2026-07-06T16:27:15.027202Z","submitted_at":"2023-10-03T17:41:46Z","title":"Spin-Spin Coupling at Small $x$: Worm-Gear and Pretzelosity TMDs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02231","snapshot_observed_at":"2026-08-06T15:14:16.713331Z","title":"Alignment of language agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.713331Z"},"links":{"cited_paper":"/paper/2310.02231","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:3bf92467f3ee29a7035210aafd96e97e13268c80be53a5db1370823751ab3455","observation_id":"858244ef-3017-4256-819a-15683edcdaf7","resolution":{"observed_at":"2026-08-06T15:14:16.713331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11147","last_updated":"2022-03-21T17:26:29Z","snapshot_observed_at":"2026-08-01T19:23:25.711617Z","submitted_at":"2022-03-21T17:26:29Z","title":"Teaching language models to support answers with verified quotes","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11147","snapshot_observed_at":"2026-08-06T15:14:16.679333Z","title":"Teaching language models to support answers with verified quotes.arXiv preprint arXiv:2203.11147,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.679333Z"},"links":{"cited_paper":"/paper/2203.11147","citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:15304e237d428629175e20db986a424280ddae1cedc560d10f118d0545618e5b","observation_id":"d9c4c7f2-95ab-46a5-925c-e1e001df020f","resolution":{"observed_at":"2026-08-06T15:14:16.679333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:14:17.082720Z","title":"Pebble: Feedback- efficient interactive reinforcement learning via relabeling experience and un- supervised pre-training","venue":null,"work_id":"eb1a4119-7b0e-4c87-957d-e2d1f43ad201","year":2021},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.674730Z"},"links":{"citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:13a407af8d4f9e34fabff1d53430491aa8749f08638aab61793a3c488ced8099","observation_id":"0719c205-ff88-44cc-a248-71620dc3f487","resolution":{"observed_at":"2026-08-06T15:14:17.087493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:14:17.046056Z","title":"Red button,","venue":null,"work_id":"2837bb63-1dc5-47fc-95c0-e84b6c7be0f6","year":2016},"citing_paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:16.720775Z"},"links":{"citing_paper":"/paper/2507.21131"},"observation_digest":"sha256:bd2e851698f4d95b74ae7d9d1cb6ca3793dbdddb897487a05485a59948960345","observation_id":"8029b33b-822d-4a61-b8c2-f69458cf70f9","resolution":{"observed_at":"2026-08-06T15:14:17.052660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.21131","last_updated":"2025-07-22T11:23:18Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T16:58:33.528124Z","submitted_at":"2025-07-22T11:23:18Z","title":"NPO: Learning Alignment and Meta-Alignment through Structured Human Feedback"},"reference_resolution":{"displayed":17,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":11,"verified_exact":0,"verified_fuzzy":4},"total_outbound_references":17},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 17 of 17 outbound references and 0 inbound Pith citation observations for arXiv:2507.21131."}