{"as_of":"2026-08-21T04:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4789b69bc91e1b6e0698409d606a5f109b9c91c1dc6ccd27f7b5b9c67a4bb2d7","coverage":[{"denominator":50,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T00:35:53.062182Z","state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.11727/citation-record","integrity":"/paper/2608.11727/integrity","json":"/paper/2608.11727/citation-record.json","paper":"/paper/2608.11727"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.916257Z","title":"Claude code, 2024.https://claude.com/claude-code","venue":null,"work_id":"833ee068-e0fd-4d2c-a4c7-e64ca61601c3","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.851448Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:596cb6dbd02c25fa21431ad9e16f43ea032ebd1a9fff3267f86d922bcc34cb36","observation_id":"dd50a0ba-0227-47cf-b045-7c1c0072b35d","resolution":{"observed_at":"2026-08-16T00:35:53.920647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.901391Z","title":"Models overview.https://platform.claude.com/docs/claude/docs/models-overview, 2026","venue":null,"work_id":"11c8922a-ad89-4d8e-bcdc-3de7ccb1e4d7","year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.856428Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:53b197d36163f9407c238578ef9a135beb028975bec72baabce4202f7569925c","observation_id":"7557ce4c-1c0b-48be-a577-805828f92a99","resolution":{"observed_at":"2026-08-16T00:35:53.906832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.888112Z","title":"Building effective agents","venue":null,"work_id":"29cac6d0-a346-4be2-9f23-a5a961c29c4d","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.860808Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:d79f38111b60c9f0c7893ebb76a346c4f1bf2e80cea05687df78cd2a8121b130","observation_id":"560ffd42-11fb-40ea-b414-76b7949bbc90","resolution":{"observed_at":"2026-08-16T00:35:53.892417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07982","last_updated":"2025-06-09T17:52:18Z","snapshot_observed_at":"2026-08-14T06:34:01.114459Z","submitted_at":"2025-06-09T17:52:18Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.07982","snapshot_observed_at":"2026-08-16T00:35:52.865427Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.865427Z"},"links":{"cited_paper":"/paper/2506.07982","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:6aad0c5ef12239c3b0a90dfcd6e1e4d677a3a6380ecb452b3d711ef03b60eb4b","observation_id":"f2924955-bd89-4804-b29b-550ed3a30566","resolution":{"observed_at":"2026-08-16T00:35:52.865427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.873544Z","title":"MLE-bench: Evaluating machine learning agents on machine learning engineering","venue":null,"work_id":"39e3755f-a636-439a-8473-b169ca32aeb0","year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.870160Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:b7fa01b86e2861547d3cb4f01c830c500974244d506bcb7006d58208c496a35d","observation_id":"73fe18b2-1629-4890-a28d-c9f2fc1df10e","resolution":{"observed_at":"2026-08-16T00:35:53.878606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.16941","last_updated":"2025-11-14T22:00:03Z","snapshot_observed_at":"2026-08-10T12:32:55.999823Z","submitted_at":"2025-09-21T06:28:17Z","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.16941","snapshot_observed_at":"2026-08-16T00:35:52.874574Z","title":"SWE-Bench Pro: Can AI agents solve long-horizon software engineering tasks?arXiv preprint arXiv:2509.16941, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.874574Z"},"links":{"cited_paper":"/paper/2509.16941","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:cd6e8c03ce49c4efcb4fb2d7a6f526c7e45df20d2c51834b0c96f6e48d93f11e","observation_id":"e4edd941-19be-441f-87f2-4f4d0ff8f748","resolution":{"observed_at":"2026-08-16T00:35:52.874574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:52.879593Z","title":"Datasheets for datasets.Communications of the ACM, 64(12):86–92, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.879593Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:2c9fde50da5b973c7158171f058ddf90d789b3766de45737861ff1736f2d1a06","observation_id":"e6047447-9ca6-4feb-8f1d-075c78ce9727","resolution":{"observed_at":"2026-08-16T00:35:52.879593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.851693Z","title":"Gemini 3.1 Pro: Model card.https://deepmind.google/models/model-cards/ gemini-3-1-pro/, 2026","venue":null,"work_id":"c81d04a8-4ac7-4d61-8c09-06ed6f61d931","year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.883711Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:af90d0232357c5154e73d97ed517e555a47e8819877a817b977cb46d62af1de4","observation_id":"3b2aadf6-456c-45d5-9404-a5c67300fee5","resolution":{"observed_at":"2026-08-16T00:35:53.856111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.14878","last_updated":"2026-05-31T05:10:10Z","snapshot_observed_at":"2026-08-15T03:13:00.015257Z","submitted_at":"2026-02-16T16:10:11Z","title":"Model Context Protocol (MCP) Tool Descriptions Are Smelly! Towards Improving AI Agent Efficiency with Augmented MCP Tool Descriptions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.14878","snapshot_observed_at":"2026-08-16T00:35:52.887791Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.887791Z"},"links":{"cited_paper":"/paper/2602.14878","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:1626e270c98b3053d0bd06e7e6863107f4dbfba5c45d7ab56d992f3de6f11169","observation_id":"94310d55-504c-4172-8c4c-91f3f2f8d770","resolution":{"observed_at":"2026-08-16T00:35:52.887791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.20251","last_updated":"2026-05-26T09:44:14Z","snapshot_observed_at":"2026-08-17T05:14:57.279738Z","submitted_at":"2026-05-18T08:34:48Z","title":"ProcCtrlBench: Evaluating Process-Level Defects and Control Preservation in LLM Coding Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.20251","snapshot_observed_at":"2026-08-16T00:35:52.892302Z","title":"ProcCtrlBench: Evaluating process-level defects and control preservation in LLM coding agents.arXiv preprint arXiv:2605.20251, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.892302Z"},"links":{"cited_paper":"/paper/2605.20251","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:80951fd56813ee966f96f0f55f875a68787a75f0bafe3d812c8654de21e07f10","observation_id":"d262a897-5a56-4027-8122-d9875218a7ec","resolution":{"observed_at":"2026-08-16T00:35:52.892302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.15553","last_updated":"2024-11-13T04:26:13Z","snapshot_observed_at":"2026-08-16T18:18:26.938255Z","submitted_at":"2024-10-21T00:59:47Z","title":"Multi-IF: Benchmarking LLMs on Multi-Turn and Multilingual Instructions Following","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.15553","snapshot_observed_at":"2026-08-16T00:35:52.896736Z","title":"Multi-IF: Benchmarking LLMs on multi-turn and multilingual instructions following.arXiv preprint arXiv:2410.15553, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.896736Z"},"links":{"cited_paper":"/paper/2410.15553","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:abdb0791bb77c4188c85967e00efabf78c71bf81792449cfef882a0ff0a5b3aa","observation_id":"1ec3ce90-7e45-4177-ba2f-09dbbbe802f9","resolution":{"observed_at":"2026-08-16T00:35:52.896736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.838850Z","title":"MLAgentBench: Evaluating language agents on machine learning experimentation","venue":null,"work_id":"e7f7e0cf-3f7e-4219-ac8e-4425b63201b1","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.901665Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:abd88c6442c284a7430e31fd94f8b80c1b658b45e898610cb334ba9af82cb578","observation_id":"32704800-6895-4df4-93cc-8bb392eb205f","resolution":{"observed_at":"2026-08-16T00:35:53.843216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.825555Z","title":"FollowBench: A multi-level fine-grained constraints following benchmark for large language models","venue":null,"work_id":"6ce663d8-09ed-490e-acf4-3591123bb5e7","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.905780Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:c3bebbf2159220812f5039bb9818d3140c72620aeb05f80b8aeafc24956fbe98","observation_id":"8f2da929-ae61-4916-8c45-5c4111efbbbe","resolution":{"observed_at":"2026-08-16T00:35:53.829881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.812428Z","title":"Jimenez, John Yang, Alexander Wettig, Shunyu Yao, Kexin Pei, Ofir Press, and Karthik Narasimhan","venue":null,"work_id":"07743d5f-3bf6-4d0c-8ff8-2dff81e6d7da","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.909830Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:28bed1d92691a1f45cb44ca31821903bc5f0fb3bebd8f2990434045c5213b619","observation_id":"6b06da15-8e42-4bf1-8756-504e09057943","resolution":{"observed_at":"2026-08-16T00:35:53.816783Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.798566Z","title":"AgentBench: Evaluating LLMs as agents","venue":null,"work_id":"f5646168-e3d7-45d4-81a0-915c722b0d45","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.913841Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:fe54df2d50fd6851de467666af102bec8acb134e9188f03fbfca6f387a736793","observation_id":"28198d65-8b6a-4848-b2d1-a9e0f1108062","resolution":{"observed_at":"2026-08-16T00:35:53.803265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-08-20T02:32:10.164015Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-16T00:35:52.917646Z","title":"Merrill, Alexander G","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.917646Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:0ce800292c8c8fb54eb4ab852c3e7d0a9383f70c1f94f5dc9080bba8ea0ce59f","observation_id":"df2458e0-0b6a-4ac4-9ef9-0fa7cf30241d","resolution":{"observed_at":"2026-08-16T00:35:52.917646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.785377Z","title":"GAIA: A benchmark for general AI assistants","venue":null,"work_id":"cd47a6cd-a556-4b6b-8f01-51e89f9e4c45","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.922037Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:d9929ff0e11b52024a02ca97558b1332bb7f4c8ec2b7d8988803579d558fb5fd","observation_id":"efb42d1a-3422-4ce0-9941-37f27846325d","resolution":{"observed_at":"2026-08-16T00:35:53.789693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.771511Z","title":"MiniMax M2.7: Model self-improvement.https://www.minimax.io/models/text/m27, 2026","venue":null,"work_id":"c5ccecbb-f715-4694-8416-8f7863d15517","year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.925970Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:e26cdf544f77d3faa211fca18893d00c8644aaa79730ecbec509f28b2368e076","observation_id":"d3118d91-10d5-4cf4-b83f-706f44eddd52","resolution":{"observed_at":"2026-08-16T00:35:53.775883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.758553Z","title":"Kimi K2.6 model card.https://huggingface.co/moonshotai/Kimi-K2.6, 2026","venue":null,"work_id":"af76472a-c843-4dbe-810b-1986848cdcd1","year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.930082Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:c43c1bcdf827614e72ce8c4999cd862d9426e629337e8e78c5c4e807e95d3693","observation_id":"e054397b-6d58-42a5-8ac3-d116d6fe3abc","resolution":{"observed_at":"2026-08-16T00:35:53.762751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.745563Z","title":"GPT-5.5 model.https://developers.openai.com/api/docs/models/gpt-5.5/, 2026","venue":null,"work_id":"95605513-31b5-48ff-a28a-bb9a3eb50e7f","year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.934313Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:940b8b7f19772d3ec9fe9574316c8f54dff9605cc55e4a2e6575402831d441ac","observation_id":"5ae1cba8-2279-44c7-b66a-0cd602440146","resolution":{"observed_at":"2026-08-16T00:35:53.749856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.732165Z","title":"Introducing SWE-bench verified.https://openai.com/index/ introducing-swe-bench-verified/, 2024","venue":null,"work_id":"69543458-2597-42b3-8aeb-26899159da90","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.938862Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:936dcb365f4849cff27ad8d3f07b3e9de3e5b9b2e26a5347e888a8fc2dd5853e","observation_id":"7a44892d-486a-4291-9b8c-752d88af48bd","resolution":{"observed_at":"2026-08-16T00:35:53.736649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.718421Z","title":"Patil, Tianjun Zhang, Xin Wang, and Joseph E","venue":null,"work_id":"1af7bfa8-d855-47cc-870e-ed7f73cecd32","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.942970Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:43db914a81f6661cfb3d3f27d1efb2aaa93e5829b1bbd660e03d360bdacfa404","observation_id":"ff6098a0-7b02-4743-a8fd-44d32f79c60b","resolution":{"observed_at":"2026-08-16T00:35:53.722887Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.704953Z","title":"Patil, Huanzhi Mao, Fanjia Yan, Charlie Ji, Vivek Suresh, Ion Stoica, and Joseph E","venue":null,"work_id":"1df352b7-ec5e-446a-bb40-c48441e449e4","year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.946957Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:66f3663db710c3200c1d34eeafd04722b198d8aac968abdb8cce222038fad060","observation_id":"767cbce2-2a76-4483-a3df-ecf87742d934","resolution":{"observed_at":"2026-08-16T00:35:53.709443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.691319Z","title":"Generalizing verifiable instruction following","venue":null,"work_id":"726be882-c192-4154-837b-2f0a682b7aab","year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.951036Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:fb0ca6b8e0c0d0eb9a4388ef4b22b37ad46ec82746d43c15a5de9b0ec620ae62","observation_id":"e7ed8de2-7101-4010-ad4c-8a6e5f775006","resolution":{"observed_at":"2026-08-16T00:35:53.696155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.678415Z","title":"AgentIF: Bench- marking instruction following of large language models in agentic scenarios","venue":null,"work_id":"edf6bd80-ab9d-4604-8a75-df1f44337405","year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.955062Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:893160bfc92c31f46f44889a9de5ebc99c504e1dc7e6952ce2f66de2b4a2ce68","observation_id":"45da52c9-23f5-44ad-961a-597793a83213","resolution":{"observed_at":"2026-08-16T00:35:53.682836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.03601","last_updated":"2024-01-07T23:01:56Z","snapshot_observed_at":"2026-08-21T00:10:56.440464Z","submitted_at":"2024-01-07T23:01:56Z","title":"InFoBench: Evaluating Instruction Following Ability in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.03601","snapshot_observed_at":"2026-08-16T00:35:52.959112Z","title":"InfoBench: Evaluating instruction following ability in large language models.arXiv preprint arXiv:2401.03601, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.959112Z"},"links":{"cited_paper":"/paper/2401.03601","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:9e99ba6e5602d56edfe2aab49b157f83d3bf4d0fa9d2bc054faf898fdcb10990","observation_id":"87e0b0e3-8571-4ad4-9c60-ec7efb7e25f3","resolution":{"observed_at":"2026-08-16T00:35:52.959112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.665225Z","title":"Qwen3.6-Max-Preview released.https://qwen.ai/blog?id=qwen3.6-max-preview, 2026","venue":null,"work_id":"0ce636d3-a771-4d06-b19b-a8f714199ac3","year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.963336Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:e59127e11c3f22a3c5b0a115e3f0297d7bdf6017ecb3cc2e44d77abb7d1da89d","observation_id":"d094cc47-8480-4ff5-8985-7aeb92da1199","resolution":{"observed_at":"2026-08-16T00:35:53.669629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.651452Z","title":"Toolformer: Language models can teach themselves to use tools","venue":null,"work_id":"c00b6195-55b2-463b-b9ee-fb1d47577091","year":2023},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.967780Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:8a6127eb2a3ab702e59cae681e9c0e0054c25d5319296ffee70900f395de6615","observation_id":"dbbf2186-5bb2-4760-95e3-1dfd29bb93e2","resolution":{"observed_at":"2026-08-16T00:35:53.655948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.638041Z","title":"Seed2.0 model card.https://yfz.ai/Seed2.0_Model_Card.pdf, 2026","venue":null,"work_id":"b81f7210-ecc1-4f03-94a1-d204e528d39c","year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.971632Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:c868b52b04e17c3b152c741beae59ecea6f0642123fc57a026c819e65a1fe4bd","observation_id":"35d5b2e3-80c5-44ac-9dc9-57b4ad1685d0","resolution":{"observed_at":"2026-08-16T00:35:53.642324Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03267","last_updated":"2026-05-01T23:55:43Z","snapshot_observed_at":"2026-08-15T00:17:32.875866Z","submitted_at":"2025-12-19T07:05:38Z","title":"OpenAI GPT-5 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.03267","snapshot_observed_at":"2026-08-16T00:35:52.975569Z","title":"OpenAI GPT-5 System Card.arXiv preprintarXiv:2601.03267, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.975569Z"},"links":{"cited_paper":"/paper/2601.03267","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:2c9e95513099f9e87be3676592227b59ec8235b6964fa7c65f3deb1c50230863","observation_id":"90050bbd-78d8-4c3a-828c-624629e18d88","resolution":{"observed_at":"2026-08-16T00:35:52.975569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01848","last_updated":"2025-04-07T12:15:49Z","snapshot_observed_at":"2026-07-06T21:03:06.857885Z","submitted_at":"2025-04-02T15:55:24Z","title":"PaperBench: Evaluating AI's Ability to Replicate AI Research","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.01848","snapshot_observed_at":"2026-08-16T00:35:52.979863Z","title":"PaperBench: Evaluating AI’s ability to replicate AI research.arXiv preprint arXiv:2504.01848, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.979863Z"},"links":{"cited_paper":"/paper/2504.01848","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:e3bb5641435c5d13019ad857abb028d52838305915094510b3b4c445c08889f1","observation_id":"82bd7606-6097-4408-aa42-c0d2a839f38e","resolution":{"observed_at":"2026-08-16T00:35:52.979863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:52.984189Z","title":"Step 3.5 Flash: Open frontier-level intelligence with 11b active parameters","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.984189Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:1e8a1d8d8777c3ebb64b255369f2a13d32c3d48c8b6d942684254a4e59bf0142","observation_id":"4a86962c-805c-4b22-bedf-f6a827ea6503","resolution":{"observed_at":"2026-08-16T00:35:52.984189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:52.988276Z","title":"Tencent unveils Hy3 preview.https://www.tencent.com/en-us/articles/2202320.html, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.988276Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:9ad51d1d89f915059cf95543d4a6408038502c005f10fae7efe5f204cc238fae","observation_id":"64808c17-4970-4704-b91e-663890d60418","resolution":{"observed_at":"2026-08-16T00:35:52.988276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.624920Z","title":"AppWorld: A controllable world of apps and people for benchmarking interactive coding agents","venue":null,"work_id":"4cf0e713-44d0-49ab-9c78-36e972de5011","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.992293Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:ba301f960a64d0094a3605fe3a4d81315c45d7b0064b3fa5d726f3a5cbc80cb8","observation_id":"77423fd4-60a9-4daf-8854-e3c86d5d6246","resolution":{"observed_at":"2026-08-16T00:35:53.629240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13208","last_updated":"2024-04-19T22:55:23Z","snapshot_observed_at":"2026-08-11T23:45:02.178667Z","submitted_at":"2024-04-19T22:55:23Z","title":"The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13208","snapshot_observed_at":"2026-08-16T00:35:52.996449Z","title":"The instruction hierarchy: Training LLMs to prioritize privileged instructions.arXiv preprint arXiv:2404.13208, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:52.996449Z"},"links":{"cited_paper":"/paper/2404.13208","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:611c760aded172bc57678171879f979478701e2e1f44bab3f885980075e57c56","observation_id":"2e12bf06-e39b-43d8-8e1e-5e0da8119158","resolution":{"observed_at":"2026-08-16T00:35:52.996449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.001057Z","title":"CodeIF-Bench: Evaluat- ing instruction-following capabilities of large language models in interactive code generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.001057Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:4ff5b9edabbd80e741c7da558ae440ad7590ada88ba5ec1866cf398ffc0a60f6","observation_id":"ee682b72-a39b-4e0c-9b19-2a99d4a285e7","resolution":{"observed_at":"2026-08-16T00:35:53.001057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.611282Z","title":"Benchmarking complex instruction- following with multiple constraints composition","venue":null,"work_id":"e2e1fffb-bc0c-4b00-bfc1-479644ec17ff","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.005172Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:8bd41656ee423c6900a8d52d3b9a6575120bb6373478b6e40c3e7d689b93a1f4","observation_id":"45288d20-abf6-4f03-a50f-de7b520a4308","resolution":{"observed_at":"2026-08-16T00:35:53.615614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.597379Z","title":"LIFBench: Evaluating the instruction following performance and stability of large language models in long- context scenarios","venue":null,"work_id":"66126c30-1b5f-4c7d-91f2-fc5571e91cc5","year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.009595Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:dcdbef413c7bf716b1090d4f734a159e305a9f517a4414ef4251f55672ed4e13","observation_id":"aae303cf-a1b7-4cc0-bbe1-371988a61791","resolution":{"observed_at":"2026-08-16T00:35:53.602044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07972","last_updated":"2024-05-30T08:55:12Z","snapshot_observed_at":"2026-08-14T22:26:00.902198Z","submitted_at":"2024-04-11T17:56:05Z","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07972","snapshot_observed_at":"2026-08-16T00:35:53.013801Z","title":"OSWorld: Benchmarking multimodal agents for open-ended tasks in real computer environments","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.013801Z"},"links":{"cited_paper":"/paper/2404.07972","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:a9e46c23a93fccb4733bf158b70fd6f23661f620ef3d2343b4a7312bbed275be","observation_id":"037af9e7-2a24-4442-a52b-9270db7beb63","resolution":{"observed_at":"2026-08-16T00:35:53.013801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.15793","last_updated":"2024-11-11T20:01:15Z","snapshot_observed_at":"2026-07-06T18:19:29.996982Z","submitted_at":"2024-05-06T17:41:33Z","title":"SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.15793","snapshot_observed_at":"2026-08-16T00:35:53.018025Z","title":"Jimenez, Alexander Wettig, Kilian Lieret, Shunyu Yao, Karthik Narasimhan, and Ofir Press","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.018025Z"},"links":{"cited_paper":"/paper/2405.15793","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:7ffc7aad468b872738d0dc1d0bfb3c6f0594fd8d22f7438998607209d580b591","observation_id":"87759daf-12d3-4382-96a8-f99e0c5820a9","resolution":{"observed_at":"2026-08-16T00:35:53.018025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.582472Z","title":"Jimenez, Alex L","venue":null,"work_id":"6df67f1e-fca2-4969-953a-4e08201b4628","year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.022196Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:2323eb167c2c8700ffc957b446fb64ae25e82331a94886e2fa105891ff0c8ad7","observation_id":"6592ac29-096c-42b7-8f8a-2e77a9c3a174","resolution":{"observed_at":"2026-08-16T00:35:53.587390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-17T20:31:29.818313Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-16T00:35:53.026282Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.026282Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:60107b31edf2e9e44cd1e8950d8d1ca20cbc4c9c4366680598a5a21596d52a46","observation_id":"7b207577-7c42-4392-ac3c-2ae67eeabfd9","resolution":{"observed_at":"2026-08-16T00:35:53.026282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.23657","last_updated":"2026-05-28T04:39:19Z","snapshot_observed_at":"2026-08-12T21:40:44.548283Z","submitted_at":"2026-05-22T14:09:41Z","title":"OpenSkillEval: Automatically Auditing the Open Skill Ecosystem for LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.23657","snapshot_observed_at":"2026-08-16T00:35:53.030478Z","title":"OpenSkillEval: Automatically auditing the open skill ecosystem for LLM agents.arXiv preprint arXiv:2605.23657, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.030478Z"},"links":{"cited_paper":"/paper/2605.23657","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:d083328bfdab5a671aa08d73d6d2bea5b40136cef4e096da57a8610b41090385","observation_id":"fca0b68c-e694-4d6d-a48d-84bda6132c8f","resolution":{"observed_at":"2026-08-16T00:35:53.030478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.567971Z","title":"GLM-5.1 release notes.https://docs.z.ai/release-notes/new-released, 2026","venue":null,"work_id":"b6417782-f00c-4453-8a9f-6f62ce1a971a","year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.035260Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:120fcd697e1cf2e6a8d3b108b3ec3396f38d6a0e9f86a6ded38104d4c8a746c3","observation_id":"ccd56daf-e901-4d80-a080-3f9f47707b8f","resolution":{"observed_at":"2026-08-16T00:35:53.572368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.554163Z","title":"CFBench: A comprehensive constraints- following benchmark for LLMs","venue":null,"work_id":"1689933e-adb1-48e7-a2f1-3282f4c4a06c","year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.039245Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:a5195258afbfad6931bbfb1025758a6fb03da589c7c5ede861ae88910408fccc","observation_id":"fa3a73aa-147a-4fc1-b001-5baadf511a37","resolution":{"observed_at":"2026-08-16T00:35:53.558499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.540669Z","title":"IHEval: Evaluating language models on following the instruction hierarchy","venue":null,"work_id":"5bb620c9-a807-4504-8cbb-96f0b09ba7b5","year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.044353Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:4e720967c1fb2688ab75257b0cffd2827be8fd9878262f78fa7867e4ffbe58e9","observation_id":"44cd9b76-4a7b-4d10-8b90-bf3e3b219e0c","resolution":{"observed_at":"2026-08-16T00:35:53.545016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.21384","last_updated":"2026-05-20T16:41:51Z","snapshot_observed_at":"2026-08-12T12:14:45.033803Z","submitted_at":"2026-05-20T16:41:51Z","title":"SpecBench: Measuring Reward Hacking in Long-Horizon Coding Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.21384","snapshot_observed_at":"2026-08-16T00:35:53.049011Z","title":"SpecBench: Measuring reward hacking in long-horizon coding agents.arXiv preprint arXiv:2605.21384, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.049011Z"},"links":{"cited_paper":"/paper/2605.21384","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:3b852c8bcb34132981976c00bd1f61a95f9ca78bc8ffa4476ea778b0abc4e53b","observation_id":"422a8ed1-6684-4f66-b8f2-120899eaada8","resolution":{"observed_at":"2026-08-16T00:35:53.049011Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-16T00:35:53.053616Z","title":"Instruction-following evaluation for large language models.arXiv preprint arXiv:2311.07911, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.053616Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:cb352618eec7edd5699f66af0e42f4e5e71c26a743cbac73063043cf5555de08","observation_id":"53a884e0-5fd7-4236-8f10-b025d740cf6a","resolution":{"observed_at":"2026-08-16T00:35:53.053616Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.527442Z","title":"Xu, Hao Zhu, Xuhui Zhou, Robert Lo, Abishek Sridhar, Xianyi Cheng, Tianyue Ou, Yonatan Bisk, Daniel Fried, Uri Alon, and Graham Neubig","venue":null,"work_id":"a9c2c7e2-0c23-427a-a0e2-986811d171d2","year":2024},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.057988Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:ae6fbf9a6fd0b66855c9f1b7ae3d85b0e18cd5ec196749f24e6c561927e2fa23","observation_id":"34ebd60a-8d73-4711-ba6a-41722ec77425","resolution":{"observed_at":"2026-08-16T00:35:53.531867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:35:53.512119Z","title":"Keep generated summaries compact,","venue":null,"work_id":"42980c7c-b83e-41bc-a0db-bf69004926f6","year":2025},"citing_paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:53.062182Z"},"links":{"citing_paper":"/paper/2608.11727"},"observation_digest":"sha256:bb4879b032d979e76dacbf5e789b4c45e2f7c0a72ec1f5771013fa59fbf558e5","observation_id":"0c81da32-8ed6-4150-904c-04ab39407e5a","resolution":{"observed_at":"2026-08-16T00:35:53.518154Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.11727","last_updated":"2026-08-12T07:07:57Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-18T17:47:46.661309Z","submitted_at":"2026-08-12T07:07:57Z","title":"Harness-IF: Evaluating Instruction Following Across Instruction Surfaces in Coding Agents"},"reference_resolution":{"displayed":50,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":0,"verified_fuzzy":29},"total_outbound_references":50},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 50 of 50 outbound references and 0 inbound Pith citation observations for arXiv:2608.11727."}