{"as_of":"2026-08-08T02:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:adba04888cafd6475045863c1178fa31e8555df7b99dbb03508bb53ea5b321aa","coverage":[{"denominator":129,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:08:08.453311Z","state":"measured"},{"denominator":104,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":104,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T15:28:14.401581Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-15T00:39:35.853384Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"cited_work":{"arxiv_id":"2505.16234","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16234","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lifebench: Evaluating length instruction following in large language models","venue":null,"work_id":"1c993b00-dc5b-4d1f-9fe1-8565a802c08d","year":2025},"citing_paper":{"arxiv_id":"2603.22267","last_updated":"2026-05-13T14:33:44Z","snapshot_observed_at":"2026-08-03T10:51:29.519900Z","submitted_at":"2026-03-23T17:51:40Z","title":"TiCo: Time-Controllable Spoken Dialogue Model","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-15T00:38:52.182973Z"},"links":{"cited_paper":"/paper/2505.16234","citing_paper":"/paper/2603.22267"},"observation_digest":"sha256:3040a819a8ea029a7313842505c7f1cfa73246b5f066099f21903ea7a1ec43c8","observation_id":"30905453-4f45-4e4b-b4e7-d3288f6a18c7","resolution":{"observed_at":"2026-05-15T00:39:35.854869Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"cited_work":{"arxiv_id":"2505.16234","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16234","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lifebench: Evaluating length instruction following in large language models","venue":null,"work_id":"1c993b00-dc5b-4d1f-9fe1-8565a802c08d","year":2025},"citing_paper":{"arxiv_id":"2604.27039","last_updated":"2026-07-20T23:24:02Z","snapshot_observed_at":"2026-08-02T15:18:52.956023Z","submitted_at":"2026-04-29T17:09:21Z","title":"Length Value Model: Scalable Value Pretraining for Token-Level Length Modeling","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-07T10:42:27.644514Z"},"links":{"cited_paper":"/paper/2505.16234","citing_paper":"/paper/2604.27039"},"observation_digest":"sha256:ca1888e771e3c6e37f2be331ecd552aa54390609198e838c974f6875634cb99d","observation_id":"eacdae65-3aa9-48a7-84d2-fe4f586cb22c","resolution":{"observed_at":"2026-05-12T09:31:26.140879Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.16234","snapshot_observed_at":"2026-08-02T15:28:14.401581Z","title":"Lifebench: Evaluating length instruction following in large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.27039","last_updated":"2026-07-20T23:24:02Z","snapshot_observed_at":"2026-08-02T15:18:52.956023Z","submitted_at":"2026-04-29T17:09:21Z","title":"Length Value Model: Scalable Value Pretraining for Token-Level Length Modeling","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-02T15:28:14.401581Z"},"links":{"cited_paper":"/paper/2505.16234","citing_paper":"/paper/2604.27039"},"observation_digest":"sha256:50f83a1d983b2ed8fe119835c3d522ef98db23bebd8be7219533d65e038dff6d","observation_id":"463c613c-2a32-44e3-bfe1-3cbabdb90551","resolution":{"observed_at":"2026-08-02T15:28:14.401581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.16234","snapshot_observed_at":"2026-08-01T18:03:36.857585Z","title":"LIFEBENCH: 16 Evaluating Length Instruction Following in Large Language Models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17420","last_updated":"2026-07-19T21:47:58Z","snapshot_observed_at":"2026-08-08T01:50:17.504767Z","submitted_at":"2026-07-19T21:47:58Z","title":"The Librarian Who Refused to Code: Model-Dependent Identity Enactment in LLM Code Generation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T18:03:36.857585Z"},"links":{"cited_paper":"/paper/2505.16234","citing_paper":"/paper/2607.17420"},"observation_digest":"sha256:38b18ffda69eeff28b20745eb80c23185003970e014f2f257d356c6f344795ca","observation_id":"9003ed95-f141-480b-8560-443b02ae4816","resolution":{"observed_at":"2026-08-01T18:03:36.857585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.16234/citation-record","integrity":"/paper/2505.16234/integrity","json":"/paper/2505.16234/citation-record.json","paper":"/paper/2505.16234"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:07:59.915175Z","title":"Abedi Firouzjaei","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:07:59.915175Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:aaa88d9dd90e8a3495305ebfcadfa24740807ac0e299e8c19e21591fa3350e52","observation_id":"d03fcbf4-d152-4577-8b0c-e084be42cdbe","resolution":{"observed_at":"2026-08-07T15:07:59.915175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:07:59.980923Z","title":"Alzantot, Y","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:07:59.980923Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:fe5c68794fdd53dc174d2605b4fc8863c4e2195d29472969b3fc7e7de9fa4331","observation_id":"52722d46-9a3d-4a13-9c58-5dec7b1fdc4a","resolution":{"observed_at":"2026-08-07T15:07:59.980923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.100922Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.100922Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:d8a18fa31c56322fd0a896326050b57198c970dae1f9a9b7c6e8d986f585e5c7","observation_id":"d62c8ae7-3dfe-454f-bb53-6b6025bae52f","resolution":{"observed_at":"2026-08-07T15:08:00.100922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.243342Z","title":"Claude 3.7 Sonnet and Claude Code","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.243342Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:9d98c456bbb53e86669829b246a7be366366b05a3b8221209ebebb4e6d1760c5","observation_id":"12609bb8-f46c-4f25-88c3-f655ffd5fb18","resolution":{"observed_at":"2026-08-07T15:08:00.243342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.370074Z","title":null,"venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.370074Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:7a2c69c7d5d42c46cdf168e1773389ad4273e06b8ca016d2818d425bd744937c","observation_id":"75e7b2b1-f57d-478d-8d66-009a1e343575","resolution":{"observed_at":"2026-08-07T15:08:00.370074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.482208Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.482208Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:bd9e0dd72275f081b31223a27dd45e4392cf53731ebd165b5bf817aae4d962c1","observation_id":"76e6cf04-00d6-4840-bdfa-52c5db7dd6ca","resolution":{"observed_at":"2026-08-07T15:08:00.482208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15204","last_updated":"2025-01-03T11:44:51Z","snapshot_observed_at":"2026-08-04T21:07:04.238345Z","submitted_at":"2024-12-19T18:59:17Z","title":"LongBench v2: Towards Deeper Understanding and Reasoning on Realistic Long-context Multitasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15204","snapshot_observed_at":"2026-08-07T15:08:00.616769Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.616769Z"},"links":{"cited_paper":"/paper/2412.15204","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:799db9eb7ab7c6ab1d207564f6f4be607d3b5262d15386394003c0a34c82bd42","observation_id":"f427873a-8f62-4dc0-91be-8d9997e3035f","resolution":{"observed_at":"2026-08-07T15:08:00.616769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.759419Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.759419Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:5451c6c641dcfcc8005256075dfe97b454de8ddad129485b9e4c8b44fad88cd7","observation_id":"c8073588-1d7d-4018-a79e-6ede488102e6","resolution":{"observed_at":"2026-08-07T15:08:00.759419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:00.923158Z","title":"Bordes, Y .-L","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:00.923158Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:50462ce26d45875332d86fefb3ea5c60d470f735b42ad3f55561a68bfe6efa18","observation_id":"85558c24-4962-4f40-9ab3-7f563662fba8","resolution":{"observed_at":"2026-08-07T15:08:00.923158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.022186Z","title":"Bosselut, A","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.022186Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:22e8a1b3fc4ce85363e138fb84fd33ae7db8bd4696bf909d9ba85226b01588ef","observation_id":"8a000c4b-be89-4539-878b-9cbd7123646c","resolution":{"observed_at":"2026-08-07T15:08:01.022186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.144208Z","title":"Butcher, M","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.144208Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:b40ecda63a71dbcc9ab66a84fe1eaaf8abc7a0c608e2229e8a2738f0bb0bef03","observation_id":"dee3ee11-3c31-4520-b503-0d57e76e40e9","resolution":{"observed_at":"2026-08-07T15:08:01.144208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.233200Z","title":"Doubao-1.5-Pro","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.233200Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:da8b5dc0bbd1be597c148e528402c118b6b113d42b57c27374d808e7b9ab6334","observation_id":"0cec495e-48db-4aa0-aa49-185bd1bb0555","resolution":{"observed_at":"2026-08-07T15:08:01.233200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.312710Z","title":"Chang, X","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.312710Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:3d4ceba40e00d9a60d009505e477799ff882d72f59e007ebca19f9c506e2330e","observation_id":"d83dcc59-f853-4d12-ad7f-9fb3c1bf25dc","resolution":{"observed_at":"2026-08-07T15:08:01.312710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-07T15:08:01.389348Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.389348Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:9b6f1f5e8d38f9e5f6ddc639c581301450edff0a34923eca26557add4e924f20","observation_id":"2845bc06-2ac8-469d-a358-83afe998016c","resolution":{"observed_at":"2026-08-07T15:08:01.389348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.475747Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.475747Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:19995abb94f48d699381b1015263326cd29f236c8793ac1eb91f68f3ad69fd9f","observation_id":"3a2e6bd6-f44e-4228-b0da-817dd6009ed0","resolution":{"observed_at":"2026-08-07T15:08:01.475747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.576145Z","title":"Chen and C","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.576145Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:c619020c40af1e52b39dc2a198aeb74abdc1e74ebb534d4af9b4a9a676efa87a","observation_id":"1b9c43a9-c31c-4586-b9f3-67543687f7d8","resolution":{"observed_at":"2026-08-07T15:08:01.576145Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.643294Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.643294Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:943027ea155b38572b53e1b35b76b9a88612ab6ea7aa1049a029534aff44ea12","observation_id":"7b252921-98f3-457a-b125-ee69f5ac761e","resolution":{"observed_at":"2026-08-07T15:08:01.643294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.723680Z","title":"Chiang, L","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.723680Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:3d2db80481dd2e3572e72019b1776b69562472de012bab354440f317afc920cc","observation_id":"f5bcecf5-7db7-4429-9b1b-7a332a67e84a","resolution":{"observed_at":"2026-08-07T15:08:01.723680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:01.821582Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.821582Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:eb5e137d89e6e7e065eedf072225ac0a1a18eb0a7569ffa4594c876f6702204b","observation_id":"9f64694b-b587-4229-9e0e-87e2b99537da","resolution":{"observed_at":"2026-08-07T15:08:01.821582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T15:08:01.907072Z","title":"Cobbe, V","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:01.907072Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:6fc84b730792e982dfa329decd1c912d4bf307395c43770072b02d64f4bdf2a1","observation_id":"9ee681a9-1b69-4f20-84b7-9353dea5e07a","resolution":{"observed_at":"2026-08-07T15:08:01.907072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.090683Z","title":"Cohan, F","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.090683Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:c6e73e69419fb196b6eb5300b93eaf9e431f738feae1a41370268b4a70baffc9","observation_id":"5eef2f6c-3b9a-4e35-bcc9-6219b1c0f537","resolution":{"observed_at":"2026-08-07T15:08:02.090683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.249374Z","title":"Collobert, J","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.249374Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:aeda88dc46ddfde142f5e2b0bdc123cecd726224733c5727a1b5e94340721264","observation_id":"b8a754b4-d26d-4973-8e81-433936305d43","resolution":{"observed_at":"2026-08-07T15:08:02.249374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.08268","last_updated":"2025-07-09T17:25:55Z","snapshot_observed_at":"2026-07-06T20:05:10.188170Z","submitted_at":"2024-12-11T10:35:45Z","title":"LCFO: Long Context and Long Form Output Dataset and Benchmarking","version":3},"cited_work":{"arxiv_id":"2412.08268","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.08268","snapshot_observed_at":"2026-08-07T15:08:12.041708Z","title":"LCFO: Long Context and Long Form Output Dataset and Benchmarking","venue":"cs.CL","work_id":"a570e7c4-1c95-4972-8f96-7f4d735aefdb","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.336424Z"},"links":{"cited_paper":"/paper/2412.08268","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:77189e7091a07d615753e3285de50b40fb61bbc839390af2f8a327321c64fa58","observation_id":"642495c2-3a0b-4a7a-9b68-31d725882625","resolution":{"observed_at":"2026-08-07T15:08:12.116541Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.496551Z","title":"Davidson, D","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.496551Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:178e9fccd783544242aef588c734e2abdc8c7c06f6f461fe19c06104efe799f6","observation_id":"8575674f-54ee-4641-8746-f565c54fe5dc","resolution":{"observed_at":"2026-08-07T15:08:02.496551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.698130Z","title":null,"venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.698130Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:4399b494b1693b94ab448cfd7cc48492ffb6314729dd512a93fb7dd67fd424ce","observation_id":"264207e3-a4f7-49e7-9fea-48f19a70c967","resolution":{"observed_at":"2026-08-07T15:08:02.698130Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.802654Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.802654Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:23247e0b678bfba23ad4a273eda9cf3311ad06b189a4ea3c824233c02711ba76","observation_id":"207f2d8c-d40f-4223-8f31-f95caddd5bb1","resolution":{"observed_at":"2026-08-07T15:08:02.802654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:02.954068Z","title":"Dubois, C","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:02.954068Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:f3c5c68e96fa5a8f61fb1db1843fea486443233bd9fe835aff473a13cbb83154","observation_id":"348202ab-4617-46ca-b372-c9179b218782","resolution":{"observed_at":"2026-08-07T15:08:02.954068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.106742Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.106742Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:5e4141312a7d572d9fda7f4d282d819563744c64e891a14a63b5896a43328ca1","observation_id":"74eced30-2b84-4219-8b65-3311a6304d66","resolution":{"observed_at":"2026-08-07T15:08:03.106742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.246808Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.246808Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:792f02e2b098e2a1b214f6ad8a651b4a3947213bee659272d8e9bd345ff0d961","observation_id":"29979e30-9658-4795-aae9-fb2f03def193","resolution":{"observed_at":"2026-08-07T15:08:03.246808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.356120Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.356120Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:9afb9ce44ad0e55e752d6dd1761ac214aa990e2bc6a6c863dca451576ffec9da","observation_id":"02d05053-243d-4d2a-9694-2ef27cfb62d8","resolution":{"observed_at":"2026-08-07T15:08:03.356120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.507635Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.507635Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:17b946362618d9b5f2a60caa1cc016b746a430c1f5c35429a05bca7851fe9b5c","observation_id":"7520112c-1822-41e5-aeb0-119846f48bbb","resolution":{"observed_at":"2026-08-07T15:08:03.507635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.619331Z","title":"Foundation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.619331Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:ddab47ec18c2921a4593ef486b02dc9d807df69aa94a150bdcd55e0a140797dd","observation_id":"1f012298-b49c-4271-9aa4-65d5d67af0c8","resolution":{"observed_at":"2026-08-07T15:08:03.619331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-08-07T13:56:34.167869Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12793","snapshot_observed_at":"2026-08-07T15:08:03.761610Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.761610Z"},"links":{"cited_paper":"/paper/2406.12793","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:04b1c53d3734ea70417439c8a4ac7570a6fee5ce588e9b9527e56fb149a26b3b","observation_id":"c4b3fe12-b0fc-4900-b9c7-ee56d1e7a38c","resolution":{"observed_at":"2026-08-07T15:08:03.761610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.821979Z","title":"Gemini 2.0 Flash","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.821979Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:d964d76bf8b0328d08f1e032b066ce57a007ef0312558befb575edcc241e3c03","observation_id":"31377b83-3942-4967-b76e-687b8a78e71f","resolution":{"observed_at":"2026-08-07T15:08:03.821979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.877823Z","title":"Gemini 2.5 Pro","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.877823Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:dd325787079f8faaf1773f1f16d57d056ac7ce645f177ee80f1b660ce69e2b09","observation_id":"ed19b53f-3d7b-4290-8c8b-879e4119f478","resolution":{"observed_at":"2026-08-07T15:08:03.877823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T15:08:03.923252Z","title":"Grattafiori, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.923252Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:93784cdbdb765366ffd2123c7017fa1673885fef17ebbce01dcd5ee36ac6563f","observation_id":"9e937a04-a837-4446-a22f-80a85be20f3b","resolution":{"observed_at":"2026-08-07T15:08:03.923252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:03.998056Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.998056Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:2fa22245503d6e1bc8bc22a64c2e944bb90c1be1604bcd48fe00386c79cb62ab","observation_id":"f19c070b-17d7-4cfa-9a32-0a7e8985ddd5","resolution":{"observed_at":"2026-08-07T15:08:03.998056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15594","last_updated":"2025-10-19T10:32:43Z","snapshot_observed_at":"2026-08-02T10:23:50.881300Z","submitted_at":"2024-11-23T16:03:35Z","title":"A Survey on LLM-as-a-Judge","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15594","snapshot_observed_at":"2026-08-07T15:08:04.044070Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.044070Z"},"links":{"cited_paper":"/paper/2411.15594","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:9c54ee82510e44813bc065069f503efda801834837f1587fceb641870de5c5d8","observation_id":"caef362d-1e6e-4407-81dc-227fe3ba992a","resolution":{"observed_at":"2026-08-07T15:08:04.044070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14656","last_updated":"2024-12-19T09:07:38Z","snapshot_observed_at":"2026-08-02T23:52:42.480802Z","submitted_at":"2024-12-19T09:07:38Z","title":"Length Controlled Generation for Black-box LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14656","snapshot_observed_at":"2026-08-07T15:08:04.110426Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.110426Z"},"links":{"cited_paper":"/paper/2412.14656","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:dd234a037979170defcd5ea17899b7e16d54a7accd117b16d58e069b4b41434c","observation_id":"0a1bbbcf-23ea-4ed9-9ec2-9ab31402e39f","resolution":{"observed_at":"2026-08-07T15:08:04.110426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T15:08:04.157989Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.157989Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:24ed0ae80f39a360425ba42cd74b1072188224296db0924aaea8f815cf43e4c3","observation_id":"6e458285-5368-4628-a9a7-9cfa31d92340","resolution":{"observed_at":"2026-08-07T15:08:04.157989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.15553","last_updated":"2024-11-13T04:26:13Z","snapshot_observed_at":"2026-07-06T19:36:45.701678Z","submitted_at":"2024-10-21T00:59:47Z","title":"Multi-IF: Benchmarking LLMs on Multi-Turn and Multilingual Instructions Following","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.15553","snapshot_observed_at":"2026-08-07T15:08:04.207512Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.207512Z"},"links":{"cited_paper":"/paper/2410.15553","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:9096775e26bddcb1c1a58ef5eb3fe3bb73a41c92766a498cbd870a237ea6f316","observation_id":"f3ab0426-6758-4aac-bd61-241b9d68ca63","resolution":{"observed_at":"2026-08-07T15:08:04.207512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.244676Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.244676Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:1cbecca4e3b4e1d028323dbb344154250a32b1640ad38bb051f75117d7b1a4cd","observation_id":"fd5c61b4-e6f6-45aa-8607-9abb2af7ac50","resolution":{"observed_at":"2026-08-07T15:08:04.244676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.314289Z","title":"Hsieh, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.314289Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:830795d3ed96bea75b2f8df87109e9eb5a684c9e0736fb539a5f9f263632479a","observation_id":"48bc6929-6b5d-479b-b24e-28ae7ef193cd","resolution":{"observed_at":"2026-08-07T15:08:04.314289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.379241Z","title":"Huang and K","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.379241Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:7846bb4152ce907dfb0831653d16bea13210a92cebd801a09f976003491329b3","observation_id":"0e770906-3b15-4109-8d83-d733cff41d74","resolution":{"observed_at":"2026-08-07T15:08:04.379241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.438908Z","title":"Huang, X","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.438908Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:77728cca50d264cbe18a126a8b59e3b7c74570ad9807f5f3341bb98968208c67","observation_id":"eaee9ae1-cf63-48cb-84a5-8bfc7eedc978","resolution":{"observed_at":"2026-08-07T15:08:04.438908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.15777","last_updated":"2024-05-29T15:50:43Z","snapshot_observed_at":"2026-08-05T13:47:30.851052Z","submitted_at":"2024-04-24T09:55:24Z","title":"A Comprehensive Survey on Evaluating Large Language Model Applications in the Medical Industry","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.15777","snapshot_observed_at":"2026-08-07T15:08:04.497560Z","title":"Huang, K","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.497560Z"},"links":{"cited_paper":"/paper/2404.15777","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:b58b60c5cf14ccfd6001023456d9bdbacb893d312e1f0cd96c055069b6ffecbb","observation_id":"5bcac5cd-9a07-4a6b-92fc-abaa0f1faeae","resolution":{"observed_at":"2026-08-07T15:08:04.497560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.03200","last_updated":"2025-01-06T18:28:04Z","snapshot_observed_at":"2026-07-06T20:17:11.484017Z","submitted_at":"2025-01-06T18:28:04Z","title":"The FACTS Grounding Leaderboard: Benchmarking LLMs' Ability to Ground Responses to Long-Form Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.03200","snapshot_observed_at":"2026-08-07T15:08:04.558589Z","title":"Jacovi, A","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.558589Z"},"links":{"cited_paper":"/paper/2501.03200","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:60bbadad79c91609e99fdd12c24b4d6d718f82193a1c672fbb180fd3f23034c4","observation_id":"9b0ac765-2a37-4dd1-9ad1-3c987da4f752","resolution":{"observed_at":"2026-08-07T15:08:04.558589Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-07T15:08:04.604933Z","title":"Jaech, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.604933Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:208d0003539fa2ccee05be736766c1da82fc864640dc705e922844b6fbe2ba5c","observation_id":"5f1ff18e-cd0c-4fee-8eae-5582c8694fa2","resolution":{"observed_at":"2026-08-07T15:08:04.604933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.656826Z","title":"Jhamtani, V","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.656826Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:10558c647d6ba833a4ff10018150bb0a9dbf77b12c3db12ba5c0d27fd238ee0b","observation_id":"67340472-1885-4a2d-8bcb-ab51bfa796ae","resolution":{"observed_at":"2026-08-07T15:08:04.656826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.700834Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.700834Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:a5ae76d9b84eb365fb4c9516d285c00bf9a3b9c9d2d6b4adce14d8c413db7131","observation_id":"4201875f-84b9-4362-abea-85aa730b65fb","resolution":{"observed_at":"2026-08-07T15:08:04.700834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.773018Z","title":"webnovel_cn (revision 745338c), 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.773018Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:12a76d8265feb0099a5448047e7780387eb3e9d8d5345202ee12320d999a665f","observation_id":"f8f43d67-01fe-4215-b405-3477faba6465","resolution":{"observed_at":"2026-08-07T15:08:04.773018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.828592Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.828592Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:62c136685c3cd1c2229cc3542e565e9a862c400df298ee07832ef9c1fef06396","observation_id":"01265994-1f64-48b2-b6cb-afa6fad5df13","resolution":{"observed_at":"2026-08-07T15:08:04.828592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.891231Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.891231Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:2cb05e57b98320438f4d7f70db417be17275c7236cb83483d05b58e54c2b91f8","observation_id":"57ab9db9-b0ff-480c-b1f8-5985cdbab959","resolution":{"observed_at":"2026-08-07T15:08:04.891231Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:04.954876Z","title":"Koupaee and W","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:04.954876Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:f09c8884f7fb88842d028d466bc6a023093ab9085583b0cd7598ffd79bdef685","observation_id":"b05c97b9-1322-40df-8b17-02e580e2fcc5","resolution":{"observed_at":"2026-08-07T15:08:04.954876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.023199Z","title":"Kry´sci´nski, N","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.023199Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:c17e94da5fbeb341953ad42d0329badfaa32ee4ea49823d6e7288c13b8fdba41","observation_id":"0a4ae45a-d174-4dbe-82ed-da489038619a","resolution":{"observed_at":"2026-08-07T15:08:05.023199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.089583Z","title":"Kuratov, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.089583Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:d743277b77bea708428d1e0bc9d29a2ab6f42ccd315fe0f9fa19cca018c6516e","observation_id":"2ce96cb9-3ce3-4822-8fd9-f5373e7b8f16","resolution":{"observed_at":"2026-08-07T15:08:05.089583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.137728Z","title":"Lample, M","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.137728Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:d4447b9d990abd0c437eacad880a11c57fa183a709f06954c293d0c0a8e97714","observation_id":"06ebd402-d64f-402d-a331-cf01a447796f","resolution":{"observed_at":"2026-08-07T15:08:05.137728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.197847Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.197847Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:bbf0ea75b15d82236a29384c42626624c371e8db7c9d6dc0cc51ed1cc98f7378","observation_id":"f5181dd5-d25f-48fb-83d8-73d647c30973","resolution":{"observed_at":"2026-08-07T15:08:05.197847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02060","last_updated":"2024-06-12T02:46:16Z","snapshot_observed_at":"2026-07-06T17:54:39.685251Z","submitted_at":"2024-04-02T15:59:11Z","title":"Long-context LLMs Struggle with Long In-context Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02060","snapshot_observed_at":"2026-08-07T15:08:05.259340Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.259340Z"},"links":{"cited_paper":"/paper/2404.02060","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:b277ce609ea25eca8815623264104389de9a0bc573ed8e87e67d92debd91f1b7","observation_id":"49645e82-b652-4441-9933-cf3cd5d6f730","resolution":{"observed_at":"2026-08-07T15:08:05.259340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20084","last_updated":"2025-06-29T16:03:19Z","snapshot_observed_at":"2026-08-07T15:59:09.657824Z","submitted_at":"2025-04-25T16:03:50Z","title":"AI Awareness","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20084","snapshot_observed_at":"2026-08-07T15:08:05.339186Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.339186Z"},"links":{"cited_paper":"/paper/2504.20084","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:1c4067393f33d19061122f20e6d51af2d3a7445805d27b305f66ab287cf24044","observation_id":"9555014a-566b-40ee-927f-818b14f6d62b","resolution":{"observed_at":"2026-08-07T15:08:05.339186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.453724Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.453724Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:8a3edf232d735328bc37f4c61a3cee58ad8463e52655a271bae72131581b4841","observation_id":"24f626ea-adfb-44ea-94c5-37195056c79d","resolution":{"observed_at":"2026-08-07T15:08:05.453724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12599","last_updated":"2024-08-22T17:59:04Z","snapshot_observed_at":"2026-07-06T19:04:43.716629Z","submitted_at":"2024-08-22T17:59:04Z","title":"Controllable Text Generation for Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12599","snapshot_observed_at":"2026-08-07T15:08:05.544649Z","title":"Liang, H","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.544649Z"},"links":{"cited_paper":"/paper/2408.12599","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:32c7f7b0e02e36a539b64d326e6f2e6f195860a82b4502e305e7baa5d6bc9a4f","observation_id":"4d97b2c8-357a-470f-8ada-2521b0871e72","resolution":{"observed_at":"2026-08-07T15:08:05.544649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.644559Z","title":"Lightman, V","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.644559Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:28511ffb590b449d6eb76ec324e1874bf698d59c21fc6f46da2c78049e4fbf70","observation_id":"730f358e-1ce1-4107-bdd1-0a7bb8fc9a1c","resolution":{"observed_at":"2026-08-07T15:08:05.644559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.718550Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.718550Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:42d7e5468cff4f5799a3c15914f5c854642249f6f112b003161604d05440c65f","observation_id":"bda9e038-2d63-4e11-90c4-3171e0cd5aa6","resolution":{"observed_at":"2026-08-07T15:08:05.718550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.827379Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.827379Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:876f752dc717660a7376607194015087b492751da6b3e0026931d4dd351e7e76","observation_id":"2fe3fbff-a7f4-4f7c-a39c-f29a07718948","resolution":{"observed_at":"2026-08-07T15:08:05.827379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-07T15:08:05.916363Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.916363Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:2bc6a40cd4f8cb70fe51422d7caf458d0cb6f2a8212de449c01ba22653358c7a","observation_id":"9f8e5921-e614-4fe2-9239-c1d4d2eca7e8","resolution":{"observed_at":"2026-08-07T15:08:05.916363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:05.968830Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:05.968830Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:179182363ecabff7d56ce89128c628a90edccd769e7fcea8f353dc3daaf7d024","observation_id":"cd80fcd9-43ab-461a-9333-a5737480603a","resolution":{"observed_at":"2026-08-07T15:08:05.968830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:06.040972Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.040972Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:4fd1bc4183971016f8dbb369e0cfd6415bd2898e73261da799cd421467779b7a","observation_id":"a8ce4bf1-95c6-4c87-aaa5-dc73ae8255f3","resolution":{"observed_at":"2026-08-07T15:08:06.040972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.920395Z","title":null,"venue":null,"work_id":"ae3877d7-c62e-484c-9c32-efc2be9cc961","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.138421Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:8b4088a5b7adca3aed1937309bd5be5fdaceca3065c7076d0257032179675bce","observation_id":"f44b67bb-10ee-43a8-ac99-29f04933732a","resolution":{"observed_at":"2026-08-07T15:08:20.965757Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:06.237974Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.237974Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:3cd4c2a5a80c28747750f79c2f06555d28535e25aecd6016f7db470020d2b7be","observation_id":"fe4da9eb-02ca-4ea7-9620-ebff6fd0f76e","resolution":{"observed_at":"2026-08-07T15:08:06.237974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07852","last_updated":"2024-04-02T01:07:05Z","snapshot_observed_at":"2026-07-31T01:56:30.416855Z","submitted_at":"2023-09-14T16:54:34Z","title":"ExpertQA: Expert-Curated Questions and Attributed Answers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07852","snapshot_observed_at":"2026-08-07T15:08:06.311393Z","title":"Malaviya, S","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.311393Z"},"links":{"cited_paper":"/paper/2309.07852","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:05172e1bd437bd02eb4007336daf2ba40943e7d0fc90d189d44384a9d63a7aa4","observation_id":"5380913f-d36a-42ef-b2c7-db46a4ade9d3","resolution":{"observed_at":"2026-08-07T15:08:06.311393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.764924Z","title":null,"venue":null,"work_id":"541adb6f-7058-404a-a158-5287163b5823","year":2013},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.397371Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:605ff60633454a92711fe5f6cc8f2757f5fbf8bce9b3b78f497dd23d5236e5ef","observation_id":"76fa2654-0be0-4af4-8c2c-89fa5fe9345d","resolution":{"observed_at":"2026-08-07T15:08:20.823549Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.725586Z","title":"Chinesenlpcorpus","venue":null,"work_id":"4c7cfa85-7605-4ff8-8047-28a9ef5d0a5c","year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.513962Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:6d7c0e81a35b1b339e624f62d5cd6971d4e2decc927f3334c80eb9a773f17866","observation_id":"69b59cad-1a91-4d25-a19e-dc830b707bc7","resolution":{"observed_at":"2026-08-07T15:08:20.743188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.592029Z","title":"Mnbvc: Massive never-ending bt vast chinese corpus.https://github.com/esbatmop/MNBVC, 2023","venue":null,"work_id":"7af3e35a-e159-4743-ad2a-92cb561ed4af","year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.674885Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:e9f3080cb3982e32d2c0ed8b1216d75264eb22cb74003c5b19c5360b2548e028","observation_id":"24b62dbc-691b-4772-9348-0ed39bdcd484","resolution":{"observed_at":"2026-08-07T15:08:20.639920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.483502Z","title":"Mostafazadeh, N","venue":null,"work_id":"cf85df4e-0147-4954-b658-325370e4c9e6","year":2016},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.786681Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:c4562e5918023742e84bbc285816894c62c48c7f55e0dd16e35480daf6874b3f","observation_id":"1173264d-9c54-422a-93b9-da8a23ce010e","resolution":{"observed_at":"2026-08-07T15:08:20.537826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.283811Z","title":"Nallapati, B","venue":null,"work_id":"f5701cb4-2a73-469a-8aba-af4c713e1ef7","year":2016},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.882313Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:9eb9ebdc4f7dfeffc6f806a82e3677c580519c231dcef64de351dc3e5a58fe21","observation_id":"85b58e44-50d7-4733-8ca1-173a4139db81","resolution":{"observed_at":"2026-08-07T15:08:20.369749Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:20.123437Z","title":"GPT-4o mini: advancing cost-efficient intelligence","venue":null,"work_id":"d1e6b443-eaf6-4dee-b1dd-37ade7ea2405","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:06.933239Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:5c911f07297a80f66160e25f89850e4244b640f1ca9f5e1875712a56f987a7b5","observation_id":"f6c223a5-af30-4fcf-a99f-2bbb5af790b9","resolution":{"observed_at":"2026-08-07T15:08:20.180851Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:07.032956Z","title":"Hello GPT-4o.https://openai.com/index/hello-gpt-4o/, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.032956Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:0df8ec5ba36478d3b26a95b7da4c0b9429161cb0ec4f48ccb6bb77f84d8de139","observation_id":"0dd40658-09f2-4836-9104-014091f5039e","resolution":{"observed_at":"2026-08-07T15:08:07.032956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.909602Z","title":"OpenAI o1-mini: Advancing cost-efficient reasoning","venue":null,"work_id":"4d40e9fb-4bb4-41fa-a337-afa3c0abbf35","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.136231Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:7632fc566906a7fc5f1d1e1068e9111fb5b7428c03dfc2154b5085bff7512350","observation_id":"c04d3ce7-0fc1-4c56-890c-8de0610a4e57","resolution":{"observed_at":"2026-08-07T15:08:20.018525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.674829Z","title":"OpenAI o3-mini: Pushing the frontier of cost-effective reasoning","venue":null,"work_id":"d5c85db0-70e8-4967-a2a7-ab86ed05c9d0","year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.212823Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:4666629270d815cc87dc51da6ae278b6df6d80fa743c3611613665a018c2c1f5","observation_id":"a5e6597d-4497-42ea-8a4e-8e43c9b6123b","resolution":{"observed_at":"2026-08-07T15:08:19.781683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.477592Z","title":null,"venue":null,"work_id":"0fe7da37-c07b-4519-9192-7470aba49654","year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.302010Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:0d27b8caab7a8d125d2d7a3968306b26eedfae1ce4afe2d18b25365fc61db4b1","observation_id":"e2b1c336-0ac1-4c52-8c46-6131af0bed4a","resolution":{"observed_at":"2026-08-07T15:08:19.563974Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.326209Z","title":null,"venue":null,"work_id":"f0b9b1b7-6f91-4156-8469-83bf21242cf2","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.421996Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:cd85ca8cf267729f4006e7113a2c08eee47821556f69c9243fcc8cd9a3ad9d8d","observation_id":"5f1e47e0-eb8d-4b31-9ccb-2af426340b81","resolution":{"observed_at":"2026-08-07T15:08:19.417697Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.202080Z","title":null,"venue":null,"work_id":"44120d1e-2f43-4b79-965b-2d6bcc5aeaad","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.521225Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:619c7e8b838dcf33ff1203c6a7304ca7618790d934adcb94c75dcda04e53433b","observation_id":"84f16566-6a20-4804-866e-4ec8307b499e","resolution":{"observed_at":"2026-08-07T15:08:19.248048Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:19.080746Z","title":null,"venue":null,"work_id":"4a9a53ea-0e0b-4013-957d-88acd1dc44aa","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.583818Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:f23a8debf52be208aac0f7317ccb30e752ac15ff0a15b77f42fe036adb417cf5","observation_id":"5a92f655-6357-4763-aa38-2c13fc4c6f1e","resolution":{"observed_at":"2026-08-07T15:08:19.112561Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.23933","last_updated":"2024-10-31T13:47:10Z","snapshot_observed_at":"2026-08-04T20:04:41.564603Z","submitted_at":"2024-10-31T13:47:10Z","title":"Language Models can Self-Lengthen to Generate Long Texts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.23933","snapshot_observed_at":"2026-08-07T15:08:07.651762Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.651762Z"},"links":{"cited_paper":"/paper/2410.23933","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:c73eb2e0784aadb2ecbcb1577fed09e7b3d8fd8f83c568b250c2b998eff1b364","observation_id":"bb0e9050-1346-4443-b9b1-b22197d61123","resolution":{"observed_at":"2026-08-07T15:08:07.651762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.16191","last_updated":"2024-09-24T15:38:11Z","snapshot_observed_at":"2026-08-03T19:24:54.460507Z","submitted_at":"2024-09-24T15:38:11Z","title":"HelloBench: Evaluating Long Text Generation Capabilities of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.16191","snapshot_observed_at":"2026-08-07T15:08:07.698689Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.698689Z"},"links":{"cited_paper":"/paper/2409.16191","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:05708dba6add7b3a714b41453ed61e11071267400843e9ed310a50b14d4fefbc","observation_id":"27da7725-b122-4352-8b43-795015fecb70","resolution":{"observed_at":"2026-08-07T15:08:07.698689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.933537Z","title":"Radford and K","venue":null,"work_id":"cbf19545-73d7-4f8a-9641-e9b916bb1077","year":2018},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.792526Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:69a83d4f2250a36af949a42dc6c902665dcb9a22b63fb7c735c2d4765fdab63f","observation_id":"cfa6af53-cd98-4c87-8776-4bde43e9b3ee","resolution":{"observed_at":"2026-08-07T15:08:18.997721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.826205Z","title":"Rafailov, A","venue":null,"work_id":"014c20da-c7c2-4f80-97d9-195c476dab6f","year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.833724Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:5e90db0e9b78bc61a76d14f70efcb728d5b87e11cff1d382801d6469bacc8de6","observation_id":"ae6662e7-ee89-48b2-be05-5437d31a5453","resolution":{"observed_at":"2026-08-07T15:08:18.862390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.702178Z","title":null,"venue":null,"work_id":"dbd82489-c557-4c97-9813-efc06927660d","year":2015},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.872975Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:849646f0a6e82bd51b2d94bc87e33086f7029a39c184f2200ed10d8773212b44","observation_id":"91c869c8-9537-4c21-a830-09f45af784a0","resolution":{"observed_at":"2026-08-07T15:08:18.746199Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.581580Z","title":null,"venue":null,"work_id":"b8ea15a9-2243-4eb8-8632-fc31b9a311e6","year":1994},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.912824Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:af98777c80acf0b8864a8457a59a110659f6fa8408b7ce7d2df93a4e22c16c98","observation_id":"dd45e7ff-fbd0-4514-b49d-f0ff703a619c","resolution":{"observed_at":"2026-08-07T15:08:18.635380Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.440592Z","title":"Sennrich, B","venue":null,"work_id":"471fb905-0171-4282-9014-fc3ef85fbd99","year":2016},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:07.978742Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:d6d18978d14e916d89d2c5e79d894b78606ef70a0adfd8b1c3927b640ce2c83f","observation_id":"84318624-9748-4bb5-86af-f06cfb3c21e0","resolution":{"observed_at":"2026-08-07T15:08:18.512717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.281176Z","title":"Shaham, M","venue":null,"work_id":"dbca6183-aaf4-473a-b5a8-e908fc0ff5b8","year":2023},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.027546Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:cd0caeb6017defa74a1fc365c4ad205f2fe23aff703989dbc35d9856afb90d49","observation_id":"75dad7df-64c9-439f-ad22-0c65def8afee","resolution":{"observed_at":"2026-08-07T15:08:18.362523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.134419Z","title":"Socher, A","venue":null,"work_id":"0a627e23-9dbe-484d-b106-5474220acf86","year":2013},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.073224Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:a699b400173f3abf5ca1bf3994c4e1b2f5c2d4aa03484ee33e3d13c4c95a1e99","observation_id":"bc065065-870c-4664-b9d5-a5dc81738cb4","resolution":{"observed_at":"2026-08-07T15:08:18.204207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:18.023213Z","title":null,"venue":null,"work_id":"bf6e396a-3509-49f6-a32d-1d2c5a8f74bf","year":2020},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.114897Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:7b3928f18e106cc5e8aadd90af37de45c076f6b971b7009435d88190d13ea572","observation_id":"d8ffbb68-2cad-46e2-b7a1-28e802e14dff","resolution":{"observed_at":"2026-08-07T15:08:18.065546Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:08.152779Z","title":"Sutskever, O","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.152779Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:61740dc1dc0cee1add4c3c6f372b1baad10d00283f3e3972502de5e8d925bed9","observation_id":"c2cd78d9-fbae-46fa-9bd9-170b91d2a54a","resolution":{"observed_at":"2026-08-07T15:08:08.152779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:17.853739Z","title":"Talmor, J","venue":null,"work_id":"0717576b-2db4-49d9-a227-dfc66e499b61","year":2019},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.199068Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:324f99870610e4e93015701b8445a036db8410d708290f45789d5a250c94e13a","observation_id":"d8b7c369-252c-4771-ba80-ece7033421e9","resolution":{"observed_at":"2026-08-07T15:08:17.933821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:17.532461Z","title":null,"venue":null,"work_id":"6449bcaf-0fbf-4dc2-af99-a624e42553e1","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.252432Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:a9eac7221cb9a8660fd8fbb0a201f5c1bc794b033a7b1000251783d196e45152","observation_id":"aeea2fa9-9cbd-49b9-8125-dfbc067f22bc","resolution":{"observed_at":"2026-08-07T15:08:17.705498Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12665","last_updated":"2025-02-11T02:09:38Z","snapshot_observed_at":"2026-07-06T18:33:02.286642Z","submitted_at":"2024-06-18T14:35:12Z","title":"CollabStory: Multi-LLM Collaborative Story Generation and Authorship Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12665","snapshot_observed_at":"2026-08-07T15:08:08.312320Z","title":"Venkatraman, N","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.312320Z"},"links":{"cited_paper":"/paper/2406.12665","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:9951ea444984289924879fcbb1b1b30c71cd634550d26dfc1c1dbd1272d46cba","observation_id":"4b274721-20f3-4893-9441-212289a8c5cb","resolution":{"observed_at":"2026-08-07T15:08:08.312320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:08:17.269639Z","title":null,"venue":null,"work_id":"d1424e90-7ae8-478a-bdf9-d39f0cb7241b","year":2024},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.393222Z"},"links":{"citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:d6757f574e1bea09c3de737527dcecd207c8a7fed80a7bf34fd82efeeabf48bc","observation_id":"fc9de4cb-0846-4cd2-a62e-5f5ab5bb5382","resolution":{"observed_at":"2026-08-07T15:08:17.405637Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15585","last_updated":"2025-06-09T02:36:20Z","snapshot_observed_at":"2026-08-07T16:00:23.244876Z","submitted_at":"2025-04-22T05:02:49Z","title":"A Comprehensive Survey in LLM(-Agent) Full Stack Safety: Data, Training and Deployment","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.15585","snapshot_observed_at":"2026-08-07T15:08:08.453311Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","version":2},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:08.453311Z"},"links":{"cited_paper":"/paper/2504.15585","citing_paper":"/paper/2505.16234"},"observation_digest":"sha256:080d13aace081e1655911dd7ed94d55dbf65f6772dfa288b14943e04336ac395","observation_id":"617e84f1-45a8-486b-bc09-a595734d39da","resolution":{"observed_at":"2026-08-07T15:08:08.453311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.16234","last_updated":"2025-06-11T02:36:18Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T01:49:56.190848Z","submitted_at":"2025-05-22T05:08:27Z","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":86,"verified_exact":1,"verified_fuzzy":13},"total_outbound_references":129},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 100 of 129 outbound references and 4 inbound Pith citation observations for arXiv:2505.16234."}