{"as_of":"2026-08-06T18:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c141e486f66ef685e8bbadaf68c42d11c60ecb7028938fe6bb6e8c5bb158c519","coverage":[{"denominator":76,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":76,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T11:26:38.634540Z","state":"measured"},{"denominator":76,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":76,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2604.16304/citation-record","integrity":"/paper/2604.16304/integrity","json":"/paper/2604.16304/citation-record.json","paper":"/paper/2604.16304"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2001.Introduction to measurement theory","venue":null,"work_id":"65bb9d90-90b6-4c6e-9625-acda2ac67ff4","year":2001},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:8cff2e937d2e09c2f45f10c371fc3a03ad4b362b37ff5af4af4bdb2a4d03d697","observation_id":"80e341ab-260a-428b-9938-e79d6df64d11","resolution":{"observed_at":"2026-05-16T11:27:48.496686Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a84d1583-2a09-4e12-8113-682ce57f6ccb","year":2019},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:15e1e3ac2a7a767dda15710be84e660ac2e522411e297b93bc5c10b3cb2123de","observation_id":"8933a2d6-0df9-4728-9308-db5098a0cf22","resolution":{"observed_at":"2026-05-16T11:27:48.453775Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1185ec80-bc20-4479-bd69-dfc5b3a753dc","year":2015},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:accf44a6ac0affc6d20e9eaeac9463df80e1ae57c273ed042fb2f91b4e85508a","observation_id":"851e59cc-2593-4d09-8730-60a0d96bab1e","resolution":{"observed_at":"2026-05-16T11:27:48.509630Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.acl-short.20","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLMs instead of human judges? a large scale empirical study across 20 NLP evaluation tasks","venue":null,"work_id":"40aaa769-7c44-492a-af44-79eb853e4933","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:ffb86006e3899451642c0e398d1a4489c66ca23407dfc03044319cfd7fbd0179","observation_id":"cd535416-a21e-41e0-9f10-63735a5f3287","resolution":{"observed_at":"2026-05-16T11:27:47.904536Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-13T23:49:48.725168+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T23:49:48.725168+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b7d3a10d-7169-4db1-8e91-826a40ae3c53","year":2006},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:0bf949db70c22ec2d313577da33dd3346eff8647c4dd74bfc944ec93189b3c28","observation_id":"b33a07bc-e042-402b-bafd-39b36c57c02b","resolution":{"observed_at":"2026-05-16T11:27:48.507918Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14782","last_updated":"2026-05-31T00:04:33Z","snapshot_observed_at":"2026-08-02T05:44:43.939336Z","submitted_at":"2024-05-23T16:50:49Z","title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","version":3},"cited_work":{"arxiv_id":"2405.14782","doi":"10.48550/arxiv.2405.14782","metadata_source":"pith","pith_arxiv_id":"2405.14782","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","venue":"cs.CL","work_id":"47b597a2-a355-4305-b1e4-80666b394ccd","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2405.14782","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:97798fba3b6bf9e88aaca91cfff362a914afcf884071f32a8501518a6b0c06bd","observation_id":"077db636-8303-496c-aea2-f7cc50494ea7","resolution":{"observed_at":"2026-05-16T18:44:50.211576Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T04:38:45.557591+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T04:38:45.557591+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2022.Thematic Analysis: A Practical Guide","venue":null,"work_id":"06065f9b-4869-423b-9786-d60625f37d89","year":2022},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:5285271f8aa3bbd5bc458b6513f749b7ed1b076f8648558057be0bb00cfd216e","observation_id":"9372f1ff-ce68-4137-a29b-9dc82731c75a","resolution":{"observed_at":"2026-05-16T11:27:48.483982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/3641289","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T02:26:42.727008Z","title":"Yu, Qiang Yang, and Xing Xie","venue":"ACM Transactions on Intelligent Systems and Technology","work_id":"baa1ad58-22ae-43a4-86d6-49b6915c12d6","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:e2ab17cbc62a627947290156e4e3cb3ca66333ef467a754fc0883993fca4c193","observation_id":"8b667205-3b4a-43b0-958a-96527d2cab0c","resolution":{"observed_at":"2026-05-16T11:27:47.878496Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-17T20:22:02.067794+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-17T20:22:02.067794+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.00061","last_updated":"2021-07-07T06:07:22Z","snapshot_observed_at":"2026-07-06T11:24:46.448222Z","submitted_at":"2021-06-30T19:00:25Z","title":"All That's 'Human' Is Not Gold: Evaluating Human Evaluation of Generated Text","version":2},"cited_work":{"arxiv_id":"2107.00061","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2107.00061","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"All that’s’ human’is not gold: Evaluating human evaluation of generated text","venue":null,"work_id":"62750e43-2749-47c2-9060-ff37f832b05b","year":2021},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2107.00061","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:aea5a7e0a65f326b5d9232a4744197e6e4c618fc093916b7ce2b674b5e53db30","observation_id":"f065e130-ba71-43de-8fa9-6c8a0cfac72e","resolution":{"observed_at":"2026-05-16T11:27:48.159299Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1073/pnas.2318124121","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Collins, Albert Q","venue":"Proceedings of the National Academy of Sciences","work_id":"4bee94cc-3582-4e9d-8147-7f40d9f658b9","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:4e53e4c10ef37988db4bba58b77fb461cd108276d853e626bf64c85d3fcff735","observation_id":"b84cc13c-78a3-4706-a23b-b4c6dfb83173","resolution":{"observed_at":"2026-05-16T11:27:47.834809Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bennett, Gary Hsieh, and Sean A","venue":null,"work_id":"c354f604-baac-4719-9803-ca2cde57bd5c","year":2017},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:99b14e0bd25613ad247b09421a0235a5503e99ce72f852c9a973333bc9f2ecb2","observation_id":"ba750c34-295a-404a-831f-a388cd25c403","resolution":{"observed_at":"2026-05-16T11:27:48.488384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"7709.157715","doi":"10.1145/157709.157715","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The WyCash portfolio management system.Proceedings of the Conference on Object-Oriented Programming Systems, Languages, and Applications, OOPSLA.1992;Part F1296(October):29–30","venue":null,"work_id":"744af7fb-ddd4-431d-850b-c8024126aedc","year":1992},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:406de78289caa773d6d61c60e5c961e3b0974e8580c753b9ce5ba851b379b1ac","observation_id":"8b1fa7d3-34e7-4c4f-a915-3acf674e0c35","resolution":{"observed_at":"2026-05-16T11:27:47.864303Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-15T20:50:25.519902+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T20:50:25.519902+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a5e8ae2c-d01a-44c5-8ddf-48abc62e4f5d","year":2022},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:791b646c65204bbe66c8c223c1e7881db549f39680d5dcab529c062a1af10054","observation_id":"f5df0b8b-740b-4758-b416-7d3930930720","resolution":{"observed_at":"2026-05-16T11:27:48.490381Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8f74796f-1ea9-4535-a3d2-0ab384ec6aff","year":2021},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:d84277483072646e2d0411b6cb9092db4947a587d10d84107961174fe78d5ea0","observation_id":"c1c6f0f1-d791-4621-a60a-03844a4c333a","resolution":{"observed_at":"2026-05-16T11:27:48.492227Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1702.08608","last_updated":"2017-03-02T19:32:10Z","snapshot_observed_at":"2026-07-30T22:42:20.127902Z","submitted_at":"2017-02-28T02:19:20Z","title":"Towards A Rigorous Science of Interpretable Machine Learning","version":2},"cited_work":{"arxiv_id":"1702.08608","doi":"10.48550/arxiv.1702.08608","metadata_source":"pith","pith_arxiv_id":"1702.08608","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards A Rigorous Science of Interpretable Machine Learning","venue":"stat.ML","work_id":"45958f3f-1e35-4e8a-8ed0-e3989a6c8be5","year":2017},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/1702.08608","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:f329c5260533fa11aa51e2fd92e759c97e895dbf9731530751b22c41017d74ec","observation_id":"1800c2d2-f378-4ad2-83b4-64e245562f74","resolution":{"observed_at":"2026-05-16T11:27:48.162805Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.309026+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.309026+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.acl-","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T09:07:47.009935Z","title":"Xu, Huazuo Gao, Deli Chen, Jiashi Li, Wangding Zeng, Xingkai Yu, Y","venue":null,"work_id":"f30bdb61-5994-46dd-a4d7-2d320a1917e0","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:3d4ebcc2c2a81d52c4d259f638f848af3b1a4618cbf134fe0d0e9cd4d1f52927","observation_id":"867b3613-c244-4fde-8aea-314055d7f3ef","resolution":{"observed_at":"2026-05-16T11:27:47.845342Z","resolver_source":"doi_truncated","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-3-031-","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T05:36:48.385946Z","title":"An extension of HybridSynchAADL and its application to collaborating au- tonomous UA Vs","venue":null,"work_id":"4a7d968d-ee43-47f9-9cca-832cf129af59","year":2023},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:6db871e5fa194a6ea1d83c919f678db6c42f2c68d2edde59ea1f4e3b19631102","observation_id":"642437ac-0201-4472-a60f-bdd951fd3cf2","resolution":{"observed_at":"2026-05-16T11:27:47.874813Z","resolver_source":"doi_truncated","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"cbf327ac-65aa-45ef-9e1f-be9ea12b89db","year":2020},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:d8b6d6e02df96a1e75bfc8b68520ff43119bd288a3c9a0e78b51178785f931be","observation_id":"68611810-516a-47da-a2b2-e4c4faf8dfdb","resolution":{"observed_at":"2026-05-16T11:27:48.504268Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9477.363972","doi":"10.1145/3639477.3639720","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gallagher, Jasmine Ratchford, Tyler Brooks, Bryan P","venue":null,"work_id":"3a780697-7965-4fee-b9ad-d70cc7220228","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:3ed628d55fa42fc3ef0272a8da124e1c5c9796fcd0e038e07ceb2fcd4e7f1b9c","observation_id":"8e8082fe-c8f1-41f5-a457-ba29a69b883d","resolution":{"observed_at":"2026-05-16T11:27:47.893352Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1613/jair.1.13715","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Repairing the cracked foundation: A survey of obstacles in evaluation practices for generated text","venue":"Journal of Artificial Intelligence Research","work_id":"856aa75d-01ef-459d-9ea6-24e26b593330","year":2023},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:e1cccaa540f7b58a39bf50332da36e254519f01fe8fb5caec7ebcd2edbb30975","observation_id":"5ea49024-9aa2-41c8-929e-ea11e617409d","resolution":{"observed_at":"2026-05-16T11:27:47.867364Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"413ef965-77e2-4b19-b7e9-50ae4513ba60","year":2007},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:3444790bbbd8587bc3383b6dfee12487a84b28d9743dafac6e3de18def9ba58e","observation_id":"f31a23bd-c6bd-43b0-889f-6b740966aec8","resolution":{"observed_at":"2026-05-16T11:27:48.482170Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":"2009.03300","doi":"10.48550/arxiv.2009.03300","metadata_source":"pith","pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Massive Multitask Language Understanding","venue":"cs.CY","work_id":"e87ec49a-544b-4ec8-8991-75298c64ff5e","year":2020},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:09e393380f542184310bc95913dafa5f5a0e240b92ca640f7800e0e446c83ae9","observation_id":"dc70fb52-17e8-4e3c-a259-cdd0d716fc8c","resolution":{"observed_at":"2026-05-16T11:27:47.852132Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0605.33008","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"43123d8f-a531-4a94-bbb5-11cbe7905bbe","year":2019},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:f85fa2dc39563d543b5a842e5db9bfe1e83ae3dc862b7c7fd82805b7729f6ee8","observation_id":"2d582065-7ce6-4686-9177-cc608f00d943","resolution":{"observed_at":"2026-05-16T11:27:48.169940Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.10632","last_updated":"2025-07-30T14:35:05Z","snapshot_observed_at":"2026-08-05T00:56:11.499372Z","submitted_at":"2024-05-17T08:49:34Z","title":"Towards interactive evaluations for interaction harms in human-AI systems","version":7},"cited_work":{"arxiv_id":"2405.10632","doi":"10.48550/arxiv.2405.10632","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.10632","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"To- wards interactive evaluations for interaction harms in human-ai systems","venue":"arXiv (Cornell University)","work_id":"e3ef566f-a48c-4b97-b9b8-2a2fea581afa","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2405.10632","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:7887a24a5b9f0b4dff177e7f157e7a11351df8e010932e81b5052c00e55a4e56","observation_id":"634a7496-4a61-4b54-9ae7-01197c2b8530","resolution":{"observed_at":"2026-05-16T11:27:47.886634Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2188.344590","doi":"10.1145/3442188.3445901","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Jacobs and Hanna Wallach","venue":null,"work_id":"599a870a-c9a3-414c-b7e0-05bf67c03e73","year":2021},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:c0b16d3816df1dd4e61d7df13c653cf2772212707d8d1bf2dcbbfd6cdc1cdcc8","observation_id":"e9667621-0abd-45e1-929f-6d6bbfea9c65","resolution":{"observed_at":"2026-05-16T11:27:47.814181Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-19T15:22:09.336629+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T15:22:09.336629+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.emnlp-main.248","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","venue":null,"work_id":"cf61d957-e8b4-4b70-b1b5-ff4f7d3d8f0f","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:42a0ccd673aa6ac7cf774cf6c43cfd16ef54389f2affd2fadb4fb1bc8a164c5a","observation_id":"7677e608-811f-490f-8044-63846081f68f","resolution":{"observed_at":"2026-05-16T11:27:47.896651Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-25T10:53:48.454413+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T10:53:48.454413+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.09746","last_updated":"2024-01-05T22:09:26Z","snapshot_observed_at":"2026-07-06T14:32:37.317828Z","submitted_at":"2022-12-19T18:59:45Z","title":"Evaluating Human-Language Model Interaction","version":5},"cited_work":{"arxiv_id":"2212.09746","doi":"10.48550/arxiv.2212.09746","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.09746","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating human-language model interaction.arXiv preprint arXiv:2212.09746","venue":"arXiv (Cornell University)","work_id":"943c6c38-155c-4ce6-b179-10ea0f1bcdd6","year":2022},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2212.09746","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:444f75d12619277152712f2c83b6f43ee6869dcb3e71b516a9f5e3bed33f801f","observation_id":"2c83638c-126d-43d7-bff1-6d1e563fee52","resolution":{"observed_at":"2026-05-16T11:27:48.177470Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1ad8cea9-b609-44cb-a293-6eced4307bbd","year":2021},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:8208cf5017e80eaa955111570fbc31738c5db748afdf55f3dbc8fbbf6acf82a1","observation_id":"b3859dfb-7c22-4824-8991-0fcfd778ef64","resolution":{"observed_at":"2026-05-16T11:27:48.478434Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09110","last_updated":"2023-10-01T21:44:23Z","snapshot_observed_at":"2026-08-01T19:14:56.803459Z","submitted_at":"2022-11-16T18:51:34Z","title":"Holistic Evaluation of Language Models","version":2},"cited_work":{"arxiv_id":"2211.09110","doi":"10.1007/bf01194075","metadata_source":"pith","pith_arxiv_id":"2211.09110","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Holistic Evaluation of Language Models","venue":"cs.CL","work_id":"cc02a01e-7218-47dc-8e66-3333e7e4adec","year":2022},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2211.09110","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:e24203a2c4a39af4b1636c88eb03e388dbed48aea0fd4f3fda3e31bf6639adc6","observation_id":"bfd5ece7-7118-4f4a-8ab9-41bfc0c926ad","resolution":{"observed_at":"2026-05-16T11:27:47.839549Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03100","last_updated":"2025-01-31T14:59:17Z","snapshot_observed_at":"2026-07-06T15:38:46.970127Z","submitted_at":"2023-06-01T00:01:43Z","title":"Rethinking Model Evaluation as Narrowing the Socio-Technical Gap","version":4},"cited_work":{"arxiv_id":"2306.03100","doi":"10.48550/arxiv.2306.03100version","metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03100","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"arXiv preprint arXiv:2306.03100 , year=","venue":null,"work_id":"280e7290-6f40-43ec-97af-cdf99b029cb1","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2306.03100","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:727ce89344a1c361c331eb0f4f77a75a6ebefd372d8654a1382f8566acf04c7e","observation_id":"fe8eb688-3b1f-421d-afc1-47a3f38d2548","resolution":{"observed_at":"2026-05-16T11:27:48.140243Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5f4408ac-912d-4c53-9982-f769a3ce4afe","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:6a63f8c50ed528745fa3ee30f580ff7a2e7a00ba862deeb070db32acb691919c","observation_id":"853f6ce4-547b-4940-ae04-1c5c627994a4","resolution":{"observed_at":"2026-05-16T11:27:48.494327Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.findings-naacl.280","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"03a36733-b3c2-4e99-9c6f-084c51f2288e","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:ce9bafa7db4a9a6f631357a4d997c1edd9a425865f9110fab862f32dc59930e4","observation_id":"001f13f0-6994-424d-9679-3127905ec226","resolution":{"observed_at":"2026-05-16T11:27:47.899203Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:17:03.600265Z","title":"In: Zong, C., Xia, F., Li, W., Navigli, R","venue":null,"work_id":"8d675bdd-79ca-48d6-9163-fc17ce0e8ece","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:366d3783a97c9556b824997ce728261052d0d28504d41249c45d63f98658a54f","observation_id":"e6bcc486-f2a1-49ba-9502-266cba625410","resolution":{"observed_at":"2026-05-16T11:27:47.848306Z","resolver_source":"doi_truncated","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-07-06T17:50:18.017647Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:ee44d2f7a9a091d0747e54a9ac53608e969bf739238868d846789ea6118dd209","observation_id":"333773b1-c3c1-434b-95d2-a90fbc04dafe","resolution":{"observed_at":"2026-05-16T11:27:48.133595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vera Liao, Alexandra Olteanu, and Ziang Xiao","venue":null,"work_id":"d7777286-ec31-495c-ae8d-8fa1c42e604e","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:4ef06c132b1082108f074a1a3d2845aaf6c8146749b977d488ef48f7151517bf","observation_id":"57d23f4c-4081-458d-8ae2-068ccde76639","resolution":{"observed_at":"2026-05-16T11:27:48.474495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"4815.364495","doi":"10.1145/3644815.3644950","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Welcome your new AI teammate: On safety analysis by leashing large language mod- els","venue":null,"work_id":"ebcedf1f-a23a-444e-812c-b161c0efc569","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:032ddc3d9472acbab2024113ce9e9f520cc215df76d9626d1f58cea6d9028a76","observation_id":"27cc4289-3688-47fe-a0f3-1557c1baa35e","resolution":{"observed_at":"2026-05-16T11:27:47.871460Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"6642.2025","doi":"10.1109/cain66642.2025.00011","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Insightai: Root cause analysis in large log files with private data using large language model","venue":null,"work_id":"9fec1c02-ad1c-4592-8eef-1f1b7ee40c45","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:a8100e247b8fa4cf243039e95550a872bb4929ccd37c32583ba798ca982bdfd9","observation_id":"ac8613c3-61c6-4564-9e12-62af4a5afc04","resolution":{"observed_at":"2026-05-16T11:27:47.787309Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ff5798ea-cbac-4308-bb93-31ed22ba4fda","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:183e41740e78b79e5f833e5cbdce1af502d5920cee4e36a8d9f37a341e56416e","observation_id":"93cfc402-2e07-4269-8275-7d3c666342eb","resolution":{"observed_at":"2026-05-16T11:27:48.476622Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2025.356951","doi":"10.1109/tai.2025.3569516","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"McIntosh and Teo Susnjak and Nalin A","venue":"IEEE Transactions on Artificial Intelligence","work_id":"9e942afd-d0bf-44fb-a745-5d6e52a2e587","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:81821345f4695e11625f33f3dc5509b6b3d4afa205ccaca0ae7221c64e0ef886","observation_id":"7fb0825c-fd3e-4972-8f08-05234f5e7fa3","resolution":{"observed_at":"2026-05-16T11:27:47.826375Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-20T10:52:17.942648+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T10:52:17.942648+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"41928fbf-2500-43c9-be48-e6db19ded4da","year":1995},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:dd46c6c8e9b1d68c971967bd0e0e33dba992616a19f3c2572807c5e8f626784d","observation_id":"27063029-dd91-45fb-8585-cf3d24a224c8","resolution":{"observed_at":"2026-05-16T11:27:48.470277Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5e49cd77-4f8c-4cd1-8a78-c3788dc98b19","year":2023},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:32fe984fb1989ee193560052bafda0c8df76270beacd8d4aa01921495dd3d644","observation_id":"fbddf141-cf64-494d-948c-6e7632a1ba2f","resolution":{"observed_at":"2026-05-16T11:27:48.472442Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"6354.2025","doi":"10.1109/icse-seip66354.2025.00008","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"In2025 IEEE/ACM 47th International Conference on Software Engineering: Software Engineering in Practice (ICSE-SEIP)","venue":null,"work_id":"f974a7cc-a937-48d8-a494-7a3824c7bf56","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:78d6d2fffe2cec931fbc40b15ecbaaa7cd2b297f78b9743f35f9340f8993ef8b","observation_id":"f0848d9d-3205-4797-bc79-af2aa97ef347","resolution":{"observed_at":"2026-05-16T11:27:47.856810Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c52ca021-af2d-47ba-b0b5-6aecf05a06d1","year":2022},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:932e9077676e4fd11e80875d6e58812231d5cf9a998d460ba75ee24f8f3a4a82","observation_id":"566e65d9-87a1-48b9-901d-a80e896c55e6","resolution":{"observed_at":"2026-05-16T11:27:48.511279Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"46b6e0db-b115-4ba4-9b1b-f96d33386a83","year":2016},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:7be02dc14373ed9ab86394103122f03698baa91737327f63aeef854b9e82bdb2","observation_id":"218778fb-06fb-4b8a-941c-aa30ea9dfed1","resolution":{"observed_at":"2026-05-16T11:27:48.451590Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.03479","last_updated":"2024-07-03T19:53:47Z","snapshot_observed_at":"2026-07-06T18:41:13.986887Z","submitted_at":"2024-07-03T19:53:47Z","title":"Human-Centered Design Recommendations for LLM-as-a-Judge","version":1},"cited_work":{"arxiv_id":"2407.03479","doi":"10.48550/arxiv.2407.03479","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.03479","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Human- centered design recommendations for llm-as-a-judge","venue":"arXiv (Cornell University)","work_id":"48108d71-8e3d-4e7d-a522-05a38a73a8fe","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2407.03479","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:cf896870bc325aebe690433ca77e9f877e828e0dc754f8788e8c64cf45557a68","observation_id":"75230a94-7698-4b96-a6b1-4f25dd1cc06a","resolution":{"observed_at":"2026-05-16T11:27:48.184935Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"4311.2025","doi":"10.1109/saner64311.2025.00011","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Jianming Chang, Songqiang Chen, Chao Peng, Hao Yu, Zhiming Li, Pengfei Gao, and Tao Xie","venue":null,"work_id":"5a61c497-b62e-4ae4-8ffd-7f2fcfdbe5b9","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:af06b9926c05157000736e2b6696ece1396617502f4e6602b1225408ffddef01","observation_id":"297542bf-2184-4d8a-8a04-f51344b29ff3","resolution":{"observed_at":"2026-05-16T11:27:48.155468Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T21:49:40.938906+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T21:49:40.938906+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"1095.33728","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T07:39:39.343080Z","title":"White, Margaret Mitchell, Timnit Gebru, Ben Hutchinson, Jamila Smith-Loud, Daniel Theron, and Parker Barnes","venue":null,"work_id":"82428032-332c-4362-9c4e-104833cdf158","year":2020},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:385ddeb012fc408dbb41b208c912093fe54c31262faedee0e93b006bbd4ebcac","observation_id":"13c8994e-228b-4a65-826a-a2d2d8fdaf64","resolution":{"observed_at":"2026-05-16T11:27:48.137236Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0654.246625","doi":"10.1145/2470654.2466257","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Roedl and Erik Stolterman","venue":null,"work_id":"2231391e-288e-4f7a-a7e2-5f551ec841c8","year":2013},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:f0f13bd096409866b9fc78446e0af00736db34e6dd1d8dac298452a9ae284182","observation_id":"d346d19b-b601-4c32-abc0-d40d1b223dcc","resolution":{"observed_at":"2026-05-16T11:27:47.809588Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"1884.359714","doi":"10.1145/3571884.3597143","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The user experience of ChatGPT: Findings from a questionnaire study of early users, in: Proc","venue":null,"work_id":"661bad86-2b01-406e-a85b-c6daf683fbc0","year":2023},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:f4abacdd8ec9c0f91d0e44ccff9cbf764c146804cacc4b93a444b6bf5f59018b","observation_id":"6aecd7c5-8c64-474c-aa7e-3604f74b1899","resolution":{"observed_at":"2026-05-16T11:27:47.821769Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"7318.2024","doi":"10.1080/10447318.2024","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Emerging roles and relationships among humans and interactive ai systems.International Journal of Human–Computer Interaction, 41(17):10595–10617","venue":null,"work_id":"cf28e4f0-940e-4ef9-b998-7ca09ad08a78","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:b6f4a700b3c7d5067416adc736455178483d3e2fc890f753552033ceb2d9b7a7","observation_id":"f2e49c1e-2fa3-4470-baae-0463ec1a19ff","resolution":{"observed_at":"2026-05-16T11:27:48.166766Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-12T11:19:14.613748+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T11:19:14.613748+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12272","last_updated":"2024-04-18T15:45:27Z","snapshot_observed_at":"2026-08-05T08:58:26.872416Z","submitted_at":"2024-04-18T15:45:27Z","title":"Who Validates the Validators? Aligning LLM-Assisted Evaluation of LLM Outputs with Human Preferences","version":1},"cited_work":{"arxiv_id":"2404.12272","doi":"10.48550/arxiv.2404.12272","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.12272","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Zamfirescu-Pereira, Björn Hartmann, Aditya G","venue":"arXiv (Cornell University)","work_id":"0b581c93-e56b-415c-b1f5-1a17d8def1e4","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2404.12272","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:e9a57c3f9652438add50d4783e5e7a8e5d8d8e54e8f3e55dd1e1afde20e7a658","observation_id":"f4010a13-58e0-44f0-8430-e0057f263945","resolution":{"observed_at":"2026-05-16T11:27:48.147823Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Shergadwala, Himabindu Lakkaraju, and Krishnaram Kenthapadi","venue":null,"work_id":"3aaffc82-9b79-4b2d-8876-04e56f10a8c9","year":null},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:fdde6fab1b5d54261a46a7562df3b7f673803029cbd7aaf95a11feb0d73bfe05","observation_id":"0871d58b-932b-4134-ae2e-06c0b6dbb904","resolution":{"observed_at":"2026-05-16T11:27:48.465991Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"30f23baf-36cb-42bc-a3b8-89498511edb6","year":null},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:fc4f55bdae92c2bed4f45ccdadcec84e88dd535d89a91e0a89bfdc434dc62e22","observation_id":"dd7e5557-0e1f-4db9-87ef-50de955665f5","resolution":{"observed_at":"2026-05-16T11:27:48.467993Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.emnlp-main.543","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What factors might be causing the significant deviations in my circadian rhythm patterns over the past 30 days?","venue":null,"work_id":"0822f722-ab42-4582-acf4-337a65f25ddd","year":2023},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:48fbbb426db3fd6dddb9cee5973eb584e77292a50d2ffa35ab6dc607705c0889","observation_id":"810bb39f-09bc-441f-9959-2707bf7a9c9a","resolution":{"observed_at":"2026-05-16T11:27:47.816958Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04615","last_updated":"2023-06-12T17:51:15Z","snapshot_observed_at":"2026-07-06T13:19:12.109592Z","submitted_at":"2022-06-09T17:05:34Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","version":3},"cited_work":{"arxiv_id":"2206.04615","doi":"10.1162/tacl_a_00688","metadata_source":"pith","pith_arxiv_id":"2206.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","venue":"cs.CL","work_id":"bb63abb3-0d50-4362-b97c-b5e725b03b39","year":2022},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2206.04615","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:241d9122ed05a1d0c359cae4f6cdedbfb9fbf9529f2cc71f3b82e6ff6db0c6fe","observation_id":"09f64f09-602a-467d-b4fc-b9500cd8316f","resolution":{"observed_at":"2026-05-16T11:27:47.797554Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1038/s41746-024-01258-7","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Stolyar, Katelyn Polanska, Karleigh R","venue":"npj Digital Medicine","work_id":"21f4d060-af9e-49e7-bbaa-10062a15c70d","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:12e41961e0a6a711f8ddba6fe5524e96693fd2499be7c19d13743f64e33bb844","observation_id":"4d4c71f7-8d3b-4242-9ea4-84eec71e99a0","resolution":{"observed_at":"2026-05-16T11:27:47.881004Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1a742ed6-a898-4550-8c18-5fbd23858ff8","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:aba6ad92f8536c7c13024bdc9cefbae746aa2116c8df6ae73ec78bc81f364499","observation_id":"deb2eb5c-4aa6-4b73-a29f-4af409cb3453","resolution":{"observed_at":"2026-05-16T11:27:48.455510Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"460160cb-8457-4224-8f85-c9bcaeb059a6","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:4057b141b6d8ce1aea3e7ba64c11374f61c037bf1242d3575c2d0a5079dd5cca","observation_id":"992133ee-f6b2-402f-8aec-2bc377c8635b","resolution":{"observed_at":"2026-05-16T11:27:48.461524Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1017/s0890060424000155","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":"Artificial intelligence for engineering design analysis and manufacturing","work_id":"fd1af824-8b26-45e7-b796-75bf6ba43445","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:b128f93e16711e492dc81034979ebeb5d63c22133f3aa69c50068c0d7166405d","observation_id":"ca2c015b-5ab0-4fed-9b85-141c840b6543","resolution":{"observed_at":"2026-05-16T11:27:47.883237Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"16ea7346-cf22-4ebb-bfeb-6f655d18c16b","year":2019},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:d5e68df2f9b351b9bd06932890999239f72677a7d06047f014002a86ac32cb26","observation_id":"25e632c4-0b85-4311-a715-4bc877246762","resolution":{"observed_at":"2026-05-16T11:27:48.459656Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/w18-5446","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proceedings of the 2018","venue":null,"work_id":"24f74631-3b99-4e7e-ab15-b562e791542e","year":2018},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:a57b8e9f8b0401d36ee8030af58a5c19a96f28eb9056534f714d0e021a54ebac","observation_id":"09bc4aa2-72d8-479d-853b-731026a23d6b","resolution":{"observed_at":"2026-05-16T11:27:47.802686Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-09T07:48:42.118348+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T07:48:42.118348+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.16391","last_updated":"2026-04-28T07:39:39Z","snapshot_observed_at":"2026-08-01T03:48:12.462157Z","submitted_at":"2024-02-26T08:31:45Z","title":"Industry Practitioners Perspectives on AI Model Quality: Perceptions, Challenges, and Solutions","version":4},"cited_work":{"arxiv_id":"2402.16391","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.16391","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Industry Practitioners Perspectives on AI Model Quality: Perceptions, Challenges, and Solutions","venue":"cs.SE","work_id":"03e38e04-d4e2-44d9-9975-fd5fdc0b767c","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2402.16391","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:ac934651e5df00ac86dfca84b7d36b27317db873bbf60d7745ea40322aead4b0","observation_id":"74507c95-25d9-4b6a-a01b-f6e7661780b6","resolution":{"observed_at":"2026-05-16T11:27:48.181459Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"89950090-5c41-42d7-a638-ac8e8e431cc1","year":null},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:9437ea254794f326877fff5ab83e31a91b581c2c6a0e497527e6ecfa67108f78","observation_id":"167c89ad-4491-4e22-b6a5-37350ef5bc41","resolution":{"observed_at":"2026-05-16T11:27:48.506150Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"3904.364191","doi":"10.1145/3613904.3641913","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MacLellan","venue":null,"work_id":"bc9b7502-4c9a-4170-8c7b-59d6452e9c57","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:aed8606397f7a0bfbd43851ec330dd172633b7744207c57f3adbf390309b1a45","observation_id":"f5951944-213e-44ba-bd14-2857391ddc4b","resolution":{"observed_at":"2026-05-16T11:27:47.832098Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13940","last_updated":"2024-09-20T07:36:46Z","snapshot_observed_at":"2026-08-04T11:07:46.404048Z","submitted_at":"2024-04-22T07:32:03Z","title":"A User-Centric Multi-Intent Benchmark for Evaluating Large Language Models","version":3},"cited_work":{"arxiv_id":"2404.13940","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.13940","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2404.13940 , year =","venue":null,"work_id":"611c86b9-5930-488a-ba62-2a336bdb3a2d","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2404.13940","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:4fadaa6fffa544e401b8d762abd4ef33329e3000f08a68c0f6c14dfd3eaf6cec","observation_id":"ca0a90b5-20ac-4025-ac26-bc4addbb0079","resolution":{"observed_at":"2026-05-16T11:27:48.173597Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"93d5806c-2fbb-4ce6-a012-3feebd98c28c","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:f3f1a7398f31cb5889ca64aaad6c97b6251a4bc746c5afbc26b6c32dfd556da2","observation_id":"74fcc166-da4b-4d85-933d-994eb24b6569","resolution":{"observed_at":"2026-05-16T11:27:48.457454Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"38be6fa4-8918-4e67-a8f2-678b1bfdf45a","year":null},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:96cbf762d15d0469e1b74f89b0ccffe0eac3569b70770e7bae2756e65e8648f3","observation_id":"d78b3bbd-e1c3-46ce-aaf5-4b5b5e18a3dd","resolution":{"observed_at":"2026-05-16T11:27:48.480383Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.11986","last_updated":"2023-10-31T18:23:32Z","snapshot_observed_at":"2026-08-02T09:23:01.679296Z","submitted_at":"2023-10-18T14:13:58Z","title":"Sociotechnical Safety Evaluation of Generative AI Systems","version":2},"cited_work":{"arxiv_id":"2310.11986","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.11986","snapshot_observed_at":"2026-07-04T12:39:48.761180Z","title":"Sociotechnical safety evaluation of generative ai systems","venue":null,"work_id":"30e90b0f-5bcf-4ade-8777-57a3469d7366","year":2023},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2310.11986","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:6e5b9271eb5996beae663875d0e6b7248414ecd5564b71f2539a093f274a9357","observation_id":"f0167c0e-0463-44fa-a1b8-6bf56655e503","resolution":{"observed_at":"2026-05-16T11:27:48.144345Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"cfd372a0-151c-4249-aa66-a040f8e9f973","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:6b7a034c55fa5d641f2cceb9ad342bdccee733662c4011d17db855c980818a3a","observation_id":"e7312317-cefb-4177-bf4e-91e5db0b9a18","resolution":{"observed_at":"2026-05-16T11:27:48.498973Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.emnlp-main.676","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vera Liao","venue":null,"work_id":"59ae0b11-a32e-48fd-aa9f-0f5dda0f32ed","year":2023},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:118e90ca9d580d46079ef13419bc802daa9026a2eee07fbc5f26877384cffb77","observation_id":"fa00ad77-93ce-4e73-b336-d056a3b35f51","resolution":{"observed_at":"2026-05-16T11:27:47.901787Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1162/tacl_a_00632","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Hashimoto","venue":"Transactions of the Association for Computational Linguistics","work_id":"0cd9c991-ceb2-4893-821b-2d3de1cad326","year":2024},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:ca24e6a2a30ca88873796a368f67e9a7c478d5e3115832b01020c1780363b124","observation_id":"0510db8b-0d87-476b-8657-23546a52059e","resolution":{"observed_at":"2026-05-16T11:27:47.889515Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4d59dac2-962f-4834-9581-b3336b07d7cb","year":2023},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:6d5b8ed4d6f331d82e01a066f05fef97b1a7a93367f788afeb30debcecc29ee4","observation_id":"5eac3c4a-458e-4c6f-a953-2af68497b5d0","resolution":{"observed_at":"2026-05-16T11:27:48.463651Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:717d31a97e4709c5f862de57d2ab71455b568bd75baed17223d9175a230e6e77","observation_id":"4bac9148-1a9f-4b4c-83a0-f9f3c95bc09c","resolution":{"observed_at":"2026-05-16T11:27:47.860553Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2022.naacl-main.24","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Deconstructing NLG Evaluation: Evaluation Practices, Assumptions, and Their Implications","venue":"Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","work_id":"1782a922-7302-496c-8595-c6ab8014cb8d","year":2022},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:d5ea49b979bb9e43de8013fee7e2f1bd47ebf0102c597bb6e4fac465defc94dc","observation_id":"4a5f1225-deb4-4932-81c5-da2e02e5226a","resolution":{"observed_at":"2026-05-16T11:27:47.842619Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-06-01T23:56:34.55927+00:00","source":"crossref_status_cache"},{"observed_at":"2026-06-01T23:56:34.55927+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a8db13c3-222f-4fae-ba37-009c84632b15","year":null},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:8607caebed5ebc8b1a6b771cef07b97fca2bf44dcc8932a977b99918b4a008f8","observation_id":"274674f2-c35f-45b2-b5b7-fc26280b0466","resolution":{"observed_at":"2026-05-16T11:27:48.486248Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the Thirteenth International Conference on Learning Representations","venue":null,"work_id":"7d30145f-c572-4d8f-aefe-1682833ad9d1","year":null},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:87dee9d84530565f37b3352e1255122240c19a729b43117dec546e784653c638","observation_id":"b2324964-4818-47e1-ad9b-514c271ced54","resolution":{"observed_at":"2026-05-16T11:27:48.501760Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","latest_version":1,"primary_category":"cs.SE","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild"},"reference_resolution":{"displayed":76,"state_counts":{"malformed_identifier":8,"metadata_mismatch":8,"parse_uncertain":0,"unresolved":23,"verified_exact":31,"verified_fuzzy":6},"total_outbound_references":76},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 76 of 76 outbound references and 0 inbound Pith citation observations for arXiv:2604.16304."}