{"as_of":"2026-08-11T22:05:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:30c31ef2a17803764697011075a94d25572da8eb6367baa65399ee9309051423","coverage":[{"denominator":26,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T05:14:36.480639Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.02150/citation-record","integrity":"/paper/2508.02150/integrity","json":"/paper/2508.02150/citation-record.json","paper":"/paper/2508.02150"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.131823Z","title":"To make the instructions more complex, I want you to identify and return five atomic constraints that can be added to the seed question","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.131823Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:6c09debd339480772a3c8443a1281ecf90672a53a6f10456f639f76a0e0ed242","observation_id":"30ef7ae3-2cc4-46d8-8669-388d24235a35","resolution":{"observed_at":"2026-08-06T05:14:36.131823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.168195Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.168195Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:5a5415d0cb52b5bd55b833810b4fd13da85b2c31cc13c5e44e5d478c732eb2ac","observation_id":"36fea6dd-fb2d-48b9-9067-968e99c5cdf4","resolution":{"observed_at":"2026-08-06T05:14:36.168195Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:37.219924Z","title":"����� �������� ����������������","venue":null,"work_id":"b266089e-0e93-4df9-8870-5230640bb6e7","year":2025},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:35.964825Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:de73467bd44ea8e46a859f4174a929c36ff1dac3107cf5e904da76dd68fcc274","observation_id":"88453eb8-30b2-4c03-8bcd-bcf6610ad178","resolution":{"observed_at":"2026-08-06T05:14:37.226387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:37.202462Z","title":"����� �������� ����������������","venue":null,"work_id":"2aab0b21-f7f7-48aa-9396-8473f3ea78c4","year":2024},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:35.985981Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:ecc7b2781c7354d594522659e13975105dbbf11673f49d89a7a2c77bfd69ab97","observation_id":"070800b4-ace1-497f-b440-6ef268daf33c","resolution":{"observed_at":"2026-08-06T05:14:37.208220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:37.182416Z","title":"����� �������� ����������������","venue":null,"work_id":"321364e2-77fa-4b2f-b10a-1ea121fe0b8c","year":2024},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.009417Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:27d9afa6c93043c3e7d9c715208cd3f7f72747ed216f7797e4b4958a962784f1","observation_id":"9f26d81b-e226-42fa-809b-e6c91f2e289c","resolution":{"observed_at":"2026-08-06T05:14:37.190858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04945","last_updated":"2025-05-31T14:42:28Z","snapshot_observed_at":"2026-08-10T21:19:18.533338Z","submitted_at":"2025-01-09T03:34:07Z","title":"Step-by-Step Mastery: Enhancing Soft Constraint Following Ability of Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04945","snapshot_observed_at":"2026-08-06T05:14:36.026653Z","title":"In ����� ���������� �� �������� ��������","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.026653Z"},"links":{"cited_paper":"/paper/2501.04945","citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:f32b610888aef2aab6000121f7804560223e1471fb18c71e1cd772d13815e3fe","observation_id":"a037e7a0-6edb-40ab-9a61-51cae0ebd1bf","resolution":{"observed_at":"2026-08-06T05:14:36.026653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:37.163744Z","title":"5-thinking: Advancing superb rea- soning models with reinforcement learning","venue":null,"work_id":"bd4d27c3-be3e-4174-8afd-013ef013a37d","year":2024},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.032798Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:5688289b395be69d622550c0a0ad893b777b362a9a8218c314cb82c79f49d874","observation_id":"67020480-fe39-4fbd-813e-01d0dd2054e7","resolution":{"observed_at":"2026-08-06T05:14:37.169330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:37.144820Z","title":"����� �������� ����������������","venue":null,"work_id":"41179c16-218e-4e67-8895-cbbe5b0c4857","year":2024},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.044328Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:b5cf0070e8842e06174c5761456d318daec557e032e4618c4142ef68948f751f","observation_id":"63de9e48-05d5-465a-9c44-e346864dee00","resolution":{"observed_at":"2026-08-06T05:14:37.149630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:37.112815Z","title":"����� �������� ����������������","venue":null,"work_id":"f64752e8-28b0-4ae3-9313-885f4a744e3f","year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.060611Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:4d845f31bac34c5dd5f5870134c187d47c014348dca389a4ca50ffc183075f14","observation_id":"2c1eee46-b55f-402a-bea0-bc133bc518f5","resolution":{"observed_at":"2026-08-06T05:14:37.127153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:37.094647Z","title":"����� �������� ����������������","venue":null,"work_id":"01067273-08b5-4da5-a7f1-2567176d465d","year":2023},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.086975Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:118ede511cebab544cef30f32a3037512c771f08ac8a313e4231bc7513b6ad15","observation_id":"c35bf93f-2f11-48ec-8320-bd251713d45a","resolution":{"observed_at":"2026-08-06T05:14:37.100189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:37.047013Z","title":"joy,” “anger,","venue":null,"work_id":"563257df-4a5f-409d-bbb4-3c9f0dd0fd3b","year":2024},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.104630Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:34a9a822eb947379eed935b6e0ee268f54d9d61e61116427a13034a8ba8d9e48","observation_id":"f10955ec-320c-4dcf-88b3-31dc5b3ad4b1","resolution":{"observed_at":"2026-08-06T05:14:37.053649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.185917Z","title":"You may choose one or more constraints from the list or propose new ones if needed","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.185917Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:fb7ec6d7b1b05da51d1a143543ebf984b72cf7ed3ce5da38a3b7cf59af66130e","observation_id":"80807db4-1b09-4088-9b74-8b1f1dce4a6e","resolution":{"observed_at":"2026-08-06T05:14:36.185917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.214321Z","title":"Your task is only to generate new constraints that can be added to it","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.214321Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:6f6b3bd2164857f95fd281fc85b91887e37b1d176661bec36ec935a6b48a0a92","observation_id":"f0fe6964-45f9-4668-b253-2224dcc20fbc","resolution":{"observed_at":"2026-08-06T05:14:36.214321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.232092Z","title":"c1\": \"<first constraint>","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.232092Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:7762c874938ef8c9c4cdd3ba63f021b5960d4e8dc3c7a0fb96b1329d1c98ec13","observation_id":"f92f8690-123b-4b0e-be3b-87b7688341dd","resolution":{"observed_at":"2026-08-06T05:14:36.232092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.256221Z","title":"No explanation, no reformulated question, no analysis—only the JSON structure","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.256221Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:78905bf4579f023b946f6a5406a3e5a478ee80b9de4a004fa61ea6438819017b","observation_id":"9361bf8c-c861-48cb-bb46-f8bf52b01c33","resolution":{"observed_at":"2026-08-06T05:14:36.256221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.274838Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.274838Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:6147e6e9342d8b69b7d7469d20c8a159c7d34e559a2f9002efe6935605803779","observation_id":"1e7f2873-a75e-42d1-8e88-bdb640d3c879","resolution":{"observed_at":"2026-08-06T05:14:36.274838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.804204Z","title":null,"venue":null,"work_id":"59dc67b4-bd8c-4bbb-bcd3-bb6c8b7c0b1c","year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.314901Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:4f6bdc2abff871caa833bc94e7bcb8dcb88a2ff7c2745d84546a367fec1e2d7a","observation_id":"bf2fcc35-c469-47f5-824d-79fce6f97408","resolution":{"observed_at":"2026-08-06T05:14:36.865813Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.743266Z","title":null,"venue":null,"work_id":"d2f2e73c-904c-403b-b277-032272112e34","year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.366593Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:9cb0600a9ba75aa2710c2a9287b46f0c7109ed31b515d9fb24b1a31bfb1dbd3c","observation_id":"b66d7156-b68c-4fe7-81ac-19e21775824d","resolution":{"observed_at":"2026-08-06T05:14:36.759569Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.720948Z","title":null,"venue":null,"work_id":"c1428653-6547-4fbe-983b-d8fc93b8e367","year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.390194Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:8ddefefd119b187e71d23b85a19b8565d970b0c8fb708ea9b9a3e66295ed64b9","observation_id":"e9c347f7-e45b-4334-83c9-30fd4bc1f17a","resolution":{"observed_at":"2026-08-06T05:14:36.729126Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.701970Z","title":null,"venue":null,"work_id":"2d9a999e-891e-493a-bd3d-b0e15777f93d","year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.417236Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:b23bca1b019e2803acd13b5cca8e27219a0e50ddaa311533cd63aa8eeda0dec7","observation_id":"b174ff10-11e9-43f2-ad0e-d195c6ee65df","resolution":{"observed_at":"2026-08-06T05:14:36.706982Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.658836Z","title":null,"venue":null,"work_id":"923ffb2f-f5a6-4a93-a24d-9a2c4fdb451b","year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.455905Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:cece5dc7e98d2d3ca952926e1b4a81189714fa3666269bd943bf43c9a5c3cb65","observation_id":"2e865080-540f-4fbe-8c8a-f2986f6aa140","resolution":{"observed_at":"2026-08-06T05:14:36.671862Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.779862Z","title":"You are a meticulous assistant who precisely adheres to all explicit and implicit constraints in user instructions","venue":null,"work_id":"9a0a3b78-685c-4c0b-8ef0-bf69af701e43","year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.344774Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:07e7b9481d8b118d00f3867e9a415bda2f8194615f0d57572d57640178dd2da9","observation_id":"3bae2fd9-3e61-43c5-9537-737eb96254dc","resolution":{"observed_at":"2026-08-06T05:14:36.788801Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.627674Z","title":null,"venue":null,"work_id":"b6cd489c-93e5-417e-b0c6-3bb1c748a100","year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.465633Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:d3d7aa58745b9d9b2289e69e2928486b44e072179eaff58c60dd8ba04640cda4","observation_id":"26ba5337-378b-4f70-81c4-be680506db97","resolution":{"observed_at":"2026-08-06T05:14:36.632653Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:36.600042Z","title":"Whisker’s Quest","venue":null,"work_id":"8dd21be5-7044-4d7a-8303-86cb9a0054af","year":2025},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:36.480639Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:8e897844d6893d61e9431dfba8bba6adecab4cd708bf2633365e463e256cf098","observation_id":"bbec9262-9820-4269-a73a-bdcc7c5e36f2","resolution":{"observed_at":"2026-08-06T05:14:36.609633Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:37.237548Z","title":"����� �������� ����������������","venue":null,"work_id":"89ab9cb9-7730-4aad-9e1b-44cd09733447","year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:35.823310Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:ac1c95d51de7b279e5d413f01143b31e87182d5034837d1299c337796afeb827","observation_id":"de110678-2512-4b55-bcfb-39ab272fa84a","resolution":{"observed_at":"2026-08-06T05:14:37.243055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:14:37.254305Z","title":"In �������� �� ��� ����������� ��� ������������� ������������ ��� ����, pages 18632–18702","venue":null,"work_id":"4cd81f4a-2e26-4a35-b8b2-630f31f78f93","year":null},"citing_paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-06T05:14:35.700324Z"},"links":{"citing_paper":"/paper/2508.02150"},"observation_digest":"sha256:9c13876927c19abb5c9628e3e01d625f2cc3ecaf799e91986546e63af37ab9c5","observation_id":"8919734b-a4e7-414d-a5ab-e7bf419868f0","resolution":{"observed_at":"2026-08-06T05:14:37.259997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.02150","last_updated":"2025-08-04T07:48:59Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T10:01:54.387269Z","submitted_at":"2025-08-04T07:48:59Z","title":"Beyond the Trade-off: Self-Supervised Reinforcement Learning for Reasoning Models' Instruction Following"},"reference_resolution":{"displayed":26,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":0,"verified_fuzzy":11},"total_outbound_references":26},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 26 of 26 outbound references and 0 inbound Pith citation observations for arXiv:2508.02150."}