{"as_of":"2026-08-18T07:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d7e12dff3d850d73aaebd0531fbea85b05f825bca9a5d0e24edb8ffa5cc0a65d","coverage":[{"denominator":44,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T15:51:31.967967Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.17455/citation-record","integrity":"/paper/2509.17455/integrity","json":"/paper/2509.17455/citation-record.json","paper":"/paper/2509.17455"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.05229","last_updated":"2025-08-27T16:24:39Z","snapshot_observed_at":"2026-08-16T08:54:56.543625Z","submitted_at":"2024-10-07T17:36:37Z","title":"GSM-Symbolic: Understanding the Limitations of Mathematical Reasoning in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05229","snapshot_observed_at":"2026-08-15T15:51:31.814803Z","title":"https://arxiv.org/abs/2410.05229","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.814803Z"},"links":{"cited_paper":"/paper/2410.05229","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:dd0c5aa0981f9a13aef1775a784c71330f19f82e365061ea6403c846dbfc8fd1","observation_id":"2de25e7c-f430-4d7c-ac19-78191c13d3bb","resolution":{"observed_at":"2026-08-15T15:51:31.814803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.595274Z","title":"(eds.) Advances in Neural Information Processing Systems, vol","venue":null,"work_id":"30160794-c2c8-4c7a-8e5a-ec63d88f7087","year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.819936Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:24d12dd0dee17901332faef37387a6a6b6b3acb9912d8c970fbc7a2524beb601","observation_id":"c52771ec-87d9-4339-8ead-85f195999dd7","resolution":{"observed_at":"2026-08-15T15:51:32.598429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.585550Z","title":"https://arxiv.org/abs/2410","venue":null,"work_id":"618a4406-9fec-4329-b491-68bb1f2ee4b7","year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.823840Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:88a40b8605473f387ed7de9920539ec87fc1896917f30f9b28220d3f1604a37f","observation_id":"68ba365b-3242-4603-bf1f-99761eaacef5","resolution":{"observed_at":"2026-08-15T15:51:32.588925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.575116Z","title":"In: Proceedings of the 36th International Conference on Neural Informa- tion Processing Systems","venue":null,"work_id":"3d96ec95-aa72-4579-b1d1-c95d749a060c","year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.827897Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:7f926155cbd7f30d0e2ac2af4a5e28498c20ab4d4379e4cb1ab3b57d78f2da67","observation_id":"6a205139-13dd-478c-97da-eb06b3ed7a56","resolution":{"observed_at":"2026-08-15T15:51:32.578889Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.563132Z","title":"In: Salakhutdinov, R., Kolter, Z., Heller, K., Weller, A., Oliver, N., Scarlett, J., Berkenkamp, F","venue":null,"work_id":"0364f2b6-9a86-45ac-8750-33130df56e58","year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.831657Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:f0551da3d73bace16317b16dbe128671be9df46e291c7ea99cda267b297cca5e","observation_id":"c8a2e701-6f54-47d8-97bd-b49e93b97198","resolution":{"observed_at":"2026-08-15T15:51:32.566969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.12588","last_updated":"2023-10-23T01:27:38Z","snapshot_observed_at":"2026-08-02T13:06:11.850456Z","submitted_at":"2022-11-22T21:06:00Z","title":"Program of Thoughts Prompting: Disentangling Computation from Reasoning for Numerical Reasoning Tasks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.12588","snapshot_observed_at":"2026-08-15T15:51:31.835390Z","title":"https://arxiv.org/abs/2211.12588","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.835390Z"},"links":{"cited_paper":"/paper/2211.12588","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:6b41dc0baab740a0128e4551446af2cec57a3c9c8eeb83168dcf34dd699b669e","observation_id":"6dd43a1d-73c5-4441-9d23-8ed6acf9aa5c","resolution":{"observed_at":"2026-08-15T15:51:31.835390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.551003Z","title":"In: Proceedings of the 40th International Conference on Machine Learning","venue":null,"work_id":"af81c678-101e-41a7-ae4f-c4dbbf66b2b0","year":2023},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.839791Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:88e2912e886c563a673ed4eb205cd1175922d828a18b0a28283f907c66988d7b","observation_id":"ebcceba5-9cf2-4475-a400-ffd2161e8a33","resolution":{"observed_at":"2026-08-15T15:51:32.554546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.538976Z","title":"In: Larochelle, H., Ranzato, M., Hadsell, R., Balcan, M.F., Lin, H","venue":null,"work_id":"35218f14-02a5-414c-98c0-6fff6b77371e","year":2020},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.843649Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:424583481e69eb07a29b139ee39a4f4fec38fa17b28b2c3b86946d1922a53553","observation_id":"f699e4aa-2cd1-4edf-81ef-5f2c55807ddd","resolution":{"observed_at":"2026-08-15T15:51:32.543121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:31.847366Z","title":"Science378(6624), 1092–1097 (2022) https://doi.org/10.1126/science.abq1158","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.847366Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:bd4970061f77c314c07e1762699fbe64e7dadc0b0efb225edad2e1629adf616a","observation_id":"98d8ed54-524a-4b50-95dc-8dc92e32e738","resolution":{"observed_at":"2026-08-15T15:51:31.847366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-08-12T14:56:35.025839Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-15T15:51:31.851013Z","title":"https://arxiv.org/abs/2310.06770","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.851013Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:a9fb769e9ea4d208ec167e5e3b91eb250c470e67f1dd5a564027748f93bdbdf3","observation_id":"7d3e2b02-2f7b-4e51-888a-7a621175fb05","resolution":{"observed_at":"2026-08-15T15:51:31.851013Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:31.855060Z","title":"ACM Trans","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.855060Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:e0cd1da986fa8f210b425ada0d66a20e3766347f151c87800b41aa239fc963eb","observation_id":"631f2b20-37a9-4aaa-8871-bf497a2fec35","resolution":{"observed_at":"2026-08-15T15:51:31.855060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.527192Z","title":"In: Thirty-seventh Conference on Neural Information Processing Systems (2023)","venue":null,"work_id":"43f3ff46-2943-413c-a6b3-aefa2af992b7","year":2023},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.858624Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:ddd235d7c5aff439ca3cd3a867e9b7837939539c41f7a2de63cb9a057ae97036","observation_id":"fbc3c4a2-6e76-4f44-a75a-c51ee6025c3e","resolution":{"observed_at":"2026-08-15T15:51:32.531058Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.516276Z","title":"From Frege to Gödel: A Source Book in Mathematical Logic1931, 1–82 (1879)","venue":null,"work_id":"4ff1ee2f-17c8-4ce2-b829-4aedbeaee6c5","year":null},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.862136Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:cb7c51476ff697d556a56c1c1fee68e3691f82b87fcd1b8b72264b82ffe483bb","observation_id":"e14ebaa3-c76a-40d5-bd55-9e1614263042","resolution":{"observed_at":"2026-08-15T15:51:32.519941Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:31.865693Z","title":"Linguistics and Philosophy4(2), 159–219 (1981) https://doi.org/10.1007/bf00350139","venue":null,"work_id":null,"year":1981},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.865693Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:767a424603206c52e09944c285df98f060afb5acdbdd5f7ad3471b873bb7533d","observation_id":"340fa614-08e4-47b5-804e-505955e5ee6e","resolution":{"observed_at":"2026-08-15T15:51:31.865693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1162/tacl_a_00518","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.024576Z","title":"Transactions of the Association for Computational Linguistics 10, 1266–1284 (2022) https://doi.org/10.1162/tacl_a_00518","venue":null,"work_id":"d5c36afe-770a-4eac-94f3-b925278ce430","year":2022},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.869371Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:024a6731645d46783885fa0301335f7fc0cf5499b7c46810884de7fd509e80a0","observation_id":"4dd9e6a8-41d4-4cd8-af4c-bb911ddfcc6b","resolution":{"observed_at":"2026-08-15T15:51:32.027947Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.505399Z","title":"Edin- burgh Advanced Textbooks in Linguistics, ??? (2016)","venue":null,"work_id":"390771ad-6f3d-4215-99f0-a6cc3a5c263d","year":2016},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.873778Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:33c5aba3d6b7f959b4b826b8abf66ddb2dee4a1e83433c53466e7c9eee751ae8","observation_id":"91ec05b2-6c00-47b5-9cab-3310e76d1a87","resolution":{"observed_at":"2026-08-15T15:51:32.509085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2022.10569","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.347639Z","title":"Computer Law & Security Review46, 105696 (2022) https://doi.org/ 10.1016/j.clsr.2022.105696","venue":null,"work_id":"ac8c38a8-bf53-426d-9531-483719247a98","year":2022},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.877245Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:074b5be70e3660879f8162a59e7c0babc713a39c67be87114f3cb9a519847e7d","observation_id":"49d40da3-c386-4820-a361-a25c3f8aa011","resolution":{"observed_at":"2026-08-15T15:51:32.353164Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:31.880675Z","title":"In: Toutanova, K., Rumshisky, A., Zettlemoyer, L., Hakkani-Tur, D., Beltagy, I., Bethard, S., Cotterell, R., Chakraborty, T., Zhou, Y","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.880675Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:709cd63e4c545e7a82157f3290eff4b09755999f22a9254128ae7e2eeec96eab","observation_id":"602e1fbf-687e-43c2-b0a1-8c0d44b6f561","resolution":{"observed_at":"2026-08-15T15:51:31.880675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.07697","last_updated":"2024-03-24T06:42:47Z","snapshot_observed_at":"2026-08-18T01:26:39.620641Z","submitted_at":"2023-07-15T03:31:38Z","title":"Think-on-Graph: Deep and Responsible Reasoning of Large Language Model on Knowledge Graph","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.07697","snapshot_observed_at":"2026-08-15T15:51:31.883664Z","title":"https://arxiv.org/abs/2307.07697","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.883664Z"},"links":{"cited_paper":"/paper/2307.07697","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:d687cc67ba839b005d032ab84935a21f42584bfc731df0f98bd8da68cca7443f","observation_id":"3c9c8866-8be6-4e81-a075-0893e691af4a","resolution":{"observed_at":"2026-08-15T15:51:31.883664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:31.886681Z","title":"In: Moens, M.-F., Huang, X., Specia, L., Yih, S.W.-t","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.886681Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:6a83eee5d29dfc26eb36397bce72fb921a398dac1612fbc733fd734a1c74e16b","observation_id":"2569a87d-923a-4833-9d3b-bf5008906fcc","resolution":{"observed_at":"2026-08-15T15:51:31.886681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.findings-naacl.176","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.005647Z","title":"(eds.) Findings of the Association for Computational Linguistics: NAACL 2025, pp","venue":null,"work_id":"cad13763-ebe2-4d7c-a518-0874ce60bfbc","year":2025},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.889348Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:47a8ec99751e932570f0763cba24e448dbcef8f5e0d5386e645f5b270588a0bb","observation_id":"b630010b-8321-47a2-a9b3-aef78d127784","resolution":{"observed_at":"2026-08-15T15:51:32.009260Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04444","last_updated":"2024-11-07T05:35:55Z","snapshot_observed_at":"2026-08-16T13:02:13.193816Z","submitted_at":"2024-11-07T05:35:55Z","title":"An Empirical Study on the Potential of LLMs in Automated Software Refactoring","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04444","snapshot_observed_at":"2026-08-15T15:51:31.892074Z","title":"https: //arxiv.org/abs/2411.04444","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.892074Z"},"links":{"cited_paper":"/paper/2411.04444","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:aa567b3a2beac3cba33a1475b48a88b8795dc88447ec22b3841a72e6c3237e17","observation_id":"936836d9-2fbb-45a8-9ecd-fabc17b42395","resolution":{"observed_at":"2026-08-15T15:51:31.892074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:31.895066Z","title":"In: Companion Proceedings of the 32nd ACM International Conference on the Foundations of Software Engineering","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.895066Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:ba674659eb757528bb1606649570c1c76c22032bd14dd257b7b4fc0dd3e58309","observation_id":"5758ca97-4232-48d5-8cea-f69f8e7e73cf","resolution":{"observed_at":"2026-08-15T15:51:31.895066Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/s10515-024-00485-2","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:31.993800Z","title":"Automated Software Engg.32(1) (2025) https://doi.org/10.1007/s10515-024-00485-2","venue":null,"work_id":"fe23d809-a4af-464e-9fd9-67ca94abe076","year":2025},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.898481Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:dfa84eb87fe742497ac419c271e9c34b3fcc101dba9baccca90afdd50a13ff30","observation_id":"51d58336-5cda-474e-8f75-47b0cb2be60f","resolution":{"observed_at":"2026-08-15T15:51:31.998627Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10509","last_updated":"2023-06-23T00:59:13Z","snapshot_observed_at":"2026-08-15T17:29:02.957931Z","submitted_at":"2022-12-20T18:26:34Z","title":"Interleaving Retrieval with Chain-of-Thought Reasoning for Knowledge-Intensive Multi-Step Questions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10509","snapshot_observed_at":"2026-08-15T15:51:31.901816Z","title":"https://arxiv.org/abs/2212.10509","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.901816Z"},"links":{"cited_paper":"/paper/2212.10509","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:0bab14df6dbd960d73cf2eb3b71c9df035b929572d7d59a5f4c2b75dc251e79c","observation_id":"5a334b3d-45b7-4490-b603-10f135a8f87a","resolution":{"observed_at":"2026-08-15T15:51:31.901816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-15T15:51:31.905574Z","title":"https://arxiv.org/abs/2110.14168","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.905574Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:636965578ce63bdf112c51eaa7c6fcc36a5b44e03e08cc01c1fb27c9bb66b158","observation_id":"e319d689-5811-40d0-8158-a427b7f3ed81","resolution":{"observed_at":"2026-08-15T15:51:31.905574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.488551Z","title":null,"venue":null,"work_id":"38a23b53-2733-439f-8f4e-ca32eba95c90","year":2023},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.909223Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:229d177769b1b74780c12e47b3a927257d31e17dace73b0d31499a67f3ee4ab6","observation_id":"5c6105d8-3613-4573-9ae0-a42a773c4324","resolution":{"observed_at":"2026-08-15T15:51:32.492230Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.478149Z","title":"In: The 2023 Conference on Empirical Methods in Natural Language Processing (2023)","venue":null,"work_id":"85bbcdf5-c5d5-45a1-af3e-363774d4c11c","year":2023},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.912511Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:7d3a239d5f690db24556fdc29e007413b6b25fb55c8614aa0b92a9441ccd599e","observation_id":"5765e5eb-a4b1-4f3c-8824-880e97b5b789","resolution":{"observed_at":"2026-08-15T15:51:32.481822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05836","last_updated":"2024-04-17T04:27:10Z","snapshot_observed_at":"2026-08-18T02:21:42.373655Z","submitted_at":"2023-06-09T12:09:15Z","title":"Can Large Language Models Infer Causation from Correlation?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05836","snapshot_observed_at":"2026-08-15T15:51:31.915637Z","title":"https: //arxiv.org/abs/2306.05836","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.915637Z"},"links":{"cited_paper":"/paper/2306.05836","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:2272e1619dde24eb787c2b4e161ff1fc31873c0807ff86535787c35978016e97","observation_id":"b0ae1015-e9e0-4bda-86c5-0c35f85e40c3","resolution":{"observed_at":"2026-08-15T15:51:31.915637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09261","last_updated":"2022-10-17T17:08:26Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-17T17:08:26Z","title":"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.09261","snapshot_observed_at":"2026-08-15T15:51:31.919237Z","title":"https: //arxiv.org/abs/2210.09261","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.919237Z"},"links":{"cited_paper":"/paper/2210.09261","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:aa079b312a7174d01ac791dbb82d066cde5e22c2bbbf6d78f195a9465183cb72","observation_id":"b6e6c072-4e2f-4f6d-868f-586cfa97845b","resolution":{"observed_at":"2026-08-15T15:51:31.919237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19187","last_updated":"2025-05-06T15:11:32Z","snapshot_observed_at":"2026-08-16T12:54:59.152764Z","submitted_at":"2025-02-26T14:50:50Z","title":"BIG-Bench Extra Hard","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19187","snapshot_observed_at":"2026-08-15T15:51:31.922614Z","title":"https://arxiv.org/abs/2502.19187","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.922614Z"},"links":{"cited_paper":"/paper/2502.19187","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:dc5749bea46b4a61e093b84c9f3e224d535c7cd6f805bf231f52daec937ffb47","observation_id":"cc2eecc1-32f7-4035-8f4d-87a148b3e677","resolution":{"observed_at":"2026-08-15T15:51:31.922614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1807.02478","last_updated":"2018-07-04T02:09:06Z","snapshot_observed_at":"2026-08-16T22:22:48.493193Z","submitted_at":"2018-07-04T02:09:06Z","title":"CAIL2018: A Large-Scale Legal Dataset for Judgment Prediction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1807.02478","snapshot_observed_at":"2026-08-15T15:51:31.926275Z","title":"https://arxiv.org/abs/1807.02478","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.926275Z"},"links":{"cited_paper":"/paper/1807.02478","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:b8f88c0c0cbe5d7aa8a6a0899ef29a60763315be0cc3b3bfc7da6ee697c3fbbb","observation_id":"2aaacc5d-86e0-46d1-8f7a-4632953e5c82","resolution":{"observed_at":"2026-08-15T15:51:31.926275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.468428Z","title":"In: Proceedings of the Annual Conference of the North American Chapter of the Association for Computational Linguistics","venue":null,"work_id":"f63a62e5-4f22-4bf8-8139-70e2c5e8563d","year":2021},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.929814Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:d1401929f51bf2bdecdc980ee224cfaaf49d3d42ce44bb212b186690e09264ac","observation_id":"a4b48898-840e-42ae-bc38-13e239884b9f","resolution":{"observed_at":"2026-08-15T15:51:32.471592Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.00582","last_updated":"2022-11-01T16:57:59Z","snapshot_observed_at":"2026-08-18T06:47:43.870760Z","submitted_at":"2022-11-01T16:57:59Z","title":"ClassActionPrediction: A Challenging Benchmark for Legal Judgment Prediction of Class Action Cases in the US","version":1},"cited_work":{"arxiv_id":"2211.00582","doi":null,"metadata_source":"pith","pith_arxiv_id":"2211.00582","snapshot_observed_at":"2026-08-15T15:51:32.127007Z","title":"ClassActionPrediction: A Challenging Benchmark for Legal Judgment Prediction of Class Action Cases in the US","venue":"cs.CL","work_id":"cbabed02-514e-4155-8004-b8f4f6e6e7aa","year":2022},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.933161Z"},"links":{"cited_paper":"/paper/2211.00582","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:515d14ccc82e046e7bb13b4826c49db2f491a29731ca9b1f348ed41f44fae5c9","observation_id":"c3ad653e-6229-4898-8cee-2461dfcfbd8a","resolution":{"observed_at":"2026-08-15T15:51:32.131292Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:31.937049Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.937049Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:5a9df496441709656c5527a2fdc4521319e2d40a19050eecbcc2c4bd7727d627","observation_id":"3a649783-5533-4bee-b65f-c15ffa6a029c","resolution":{"observed_at":"2026-08-15T15:51:31.937049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.453032Z","title":"In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp","venue":null,"work_id":"9d126899-f2aa-44cc-9b8a-887afe579fb6","year":2020},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.940675Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:c409b2fc962f349090007b8e07ab066f378a4635c2bc06c3c07ee99a6e2a5290","observation_id":"75b5497b-abee-4b24-9e14-f7fa024a2853","resolution":{"observed_at":"2026-08-15T15:51:32.455996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.443271Z","title":null,"venue":null,"work_id":"45865279-b7f1-4f45-8591-057321db3049","year":2024},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.943871Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:f0f36d7f4d4e64687828e6aad5c1ec341ed96f508d0fae9fe20545081f2b61e2","observation_id":"0460b5c6-7bc2-47b0-8404-d1fe2a1bb79b","resolution":{"observed_at":"2026-08-15T15:51:32.446676Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.432436Z","title":"https://github.com/google-research/google-research/tree/master/mbpp","venue":null,"work_id":"a0aaef6e-e38a-42e0-855e-8cb60fb0600c","year":2025},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.947174Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:fb983dac5db036a9ea4a8e7d473a7ad8d83ea2502a1cbd69cc81b9df5cd8dd0b","observation_id":"4ab7248e-d1e8-4325-863b-2d4452db7ac4","resolution":{"observed_at":"2026-08-15T15:51:32.436167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.421675Z","title":"https://huggingface.co/datasets/PatrickHaller/the-stack-python-1M","venue":null,"work_id":"652a38b6-07a9-48ba-8e1e-f11132d1297e","year":2025},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.950866Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:f4364a0fba3b90f54181b82179e2538893bb1526168939c1f4d0077340ba069a","observation_id":"a82f4095-b977-48cc-9126-b7867245044c","resolution":{"observed_at":"2026-08-15T15:51:32.425791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.411343Z","title":"https://huggingface.co/datasets/notbadai/python_functions_reasoning","venue":null,"work_id":"295e2e12-c0d7-44d1-85b5-0e0f26f54b6b","year":2025},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.954341Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:cfb8e18105557fac0dd10cbf463c6ed6df8d9a09d0e69d9881786a4e47512b35","observation_id":"7c9af900-b20e-4319-b889-3c04fe19f775","resolution":{"observed_at":"2026-08-15T15:51:32.414983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04905","last_updated":"2025-03-20T03:28:56Z","snapshot_observed_at":"2026-08-16T13:02:01.818557Z","submitted_at":"2024-11-07T17:47:25Z","title":"OpenCoder: The Open Cookbook for Top-Tier Code Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04905","snapshot_observed_at":"2026-08-15T15:51:31.957576Z","title":"https://arxiv.org/abs/2411.04905 45","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.957576Z"},"links":{"cited_paper":"/paper/2411.04905","citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:7518c96396895de09f27fed19926b21f4d2d70dbedd1489296ea162fe19e6995","observation_id":"63cee2cd-60f9-4c6c-8658-d942e70e3a97","resolution":{"observed_at":"2026-08-15T15:51:31.957576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.400875Z","title":"MIT Press, ??? (2009)","venue":null,"work_id":"2c86e524-2f75-4881-b327-12487f277e58","year":2009},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.960984Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:524385c247548e4db8f67e799b617918fcbdfb554581242e10a03bc4881348e9","observation_id":"d076f12f-ff21-45e1-9944-3c955770868c","resolution":{"observed_at":"2026-08-15T15:51:32.404402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:31.964547Z","title":"IEEE Transactions on Software Engineering SE-2(4), 308–320 (1976) https://doi.org/10.1109/TSE.1976.233837","venue":null,"work_id":null,"year":1976},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.964547Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:0c9abd4b6d962b62922e2c5505e483cf27b5ba129a4e9b2e22ec9e3acd256baf","observation_id":"1b39479a-3578-4aa5-9a27-71c97cf341a4","resolution":{"observed_at":"2026-08-15T15:51:31.964547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T15:51:32.390128Z","title":"https://docs.python.org/3/ library/ast.html","venue":null,"work_id":"165a1807-95a5-40de-add6-20106c907522","year":2025},"citing_paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:31.967967Z"},"links":{"citing_paper":"/paper/2509.17455"},"observation_digest":"sha256:fa0d41d6bb04735de5e5261e957e56b4b60e77cd1f88fb7f7c32cf59450951f0","observation_id":"93d47042-3557-4b95-96d0-478cacb086b2","resolution":{"observed_at":"2026-08-15T15:51:32.393731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.17455","last_updated":"2026-06-08T04:22:45Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-18T02:22:05.106782Z","submitted_at":"2025-09-22T07:49:58Z","title":"Understanding Benchmark Language Under Weakened Formal Semantics"},"reference_resolution":{"displayed":44,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":22,"verified_exact":3,"verified_fuzzy":17},"total_outbound_references":44},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 44 of 44 outbound references and 0 inbound Pith citation observations for arXiv:2509.17455."}