{"as_of":"2026-08-08T07:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:234c4279bfab2ba8e124652821051c6fd74554be6208b245f825d186a153ed32","coverage":[{"denominator":15,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:51:32.423009Z","state":"measured"},{"denominator":15,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":15,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.00064/citation-record","integrity":"/paper/2506.00064/integrity","json":"/paper/2506.00064/citation-record.json","paper":"/paper/2506.00064"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:34.590971Z","title":null,"venue":null,"work_id":"44472143-a5ec-49df-bbb0-aa87396c7417","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.431122Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:7c1ab7a2feb9d20cdac02680f0fbe52437f3755c0d2e4683548b29845ee5964b","observation_id":"cca2bcf8-2d83-4d32-9815-1be91e2442b6","resolution":{"observed_at":"2026-08-07T12:51:34.700354Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:34.295402Z","title":"primary-category","venue":null,"work_id":"2289c282-542a-4074-bf41-89d55eb40ef5","year":2006},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.491616Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:259efa153a143b76ad1a252586499f6b13355e2a5a62c610416e79690f6eb894","observation_id":"b71a70cc-1971-4efe-b594-980cbe935a5f","resolution":{"observed_at":"2026-08-07T12:51:34.430233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"7582.3777","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:32.720039Z","title":"How long does it take to drive from the university to the beach?","venue":null,"work_id":"139dd36c-43f5-4f7a-bb23-db4513ed6b7f","year":2017},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:32.423009Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:6059c2896ac0022dc1aa9b1a9c1f73f3788a09eea69c2a0d45e7b3e406664c89","observation_id":"39511c44-53c8-4aa3-b1e6-fedb872d6332","resolution":{"observed_at":"2026-08-07T12:51:32.813447Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:34.829020Z","title":"Please ensure that the categorization is correct and not ambiguous","venue":null,"work_id":"397b3cdd-73c5-4014-862a-fe3d721e8b13","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.583967Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:6c4462f2ceba41498b5dec9fd7c88629dd4afdcc78ea5987de44e4f0a3684def","observation_id":"943478cc-e6db-48eb-ac9e-646001313517","resolution":{"observed_at":"2026-08-07T12:51:34.995301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:34.066583Z","title":null,"venue":null,"work_id":"21b85744-eaee-4f38-993e-4c7dba12e28d","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.678732Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:48c1d297cc3b1b43bb52d5654d33250d703234dc5526c43ed52aef51f5595b5f","observation_id":"7a63a5ca-9ff1-4132-90d7-65583dbac4a5","resolution":{"observed_at":"2026-08-07T12:51:34.138052Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.861820Z","title":"primary-category","venue":null,"work_id":"9fb0e94b-5b35-49ef-a71c-8c7cb952da41","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.730040Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:6c7b82ecfd0b8846c1f0f54a2ee90c2131c1646172fddcba9c59a336dde436c5","observation_id":"25bded7d-f0b8-46f8-b04d-486ab57e748c","resolution":{"observed_at":"2026-08-07T12:51:33.986171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.805703Z","title":null,"venue":null,"work_id":"65bdc23b-b8cb-4689-9d01-748ee1276f07","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.778209Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:af346ea0209f79ece319a9c92c1ffdd1a96b92f36dafae68e228c5d474f5be79","observation_id":"1a5f419e-3793-4026-af34-9ae93102be26","resolution":{"observed_at":"2026-08-07T12:51:33.848324Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.690972Z","title":"Textual context and subject specification","venue":null,"work_id":"92ebb24b-4f66-4f20-8352-3af96ebba535","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.870624Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:cb64f8e0a0cac2a9b19522fd588aec49f7f99c1c7997e36f035f9ac3c7e85d7a","observation_id":"af8b8a3d-71ee-495d-8a29-f7b751212fdb","resolution":{"observed_at":"2026-08-07T12:51:33.736624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.573706Z","title":"primary-category","venue":null,"work_id":"fddbfe9c-c0e1-44af-8ce4-ef30509da9b8","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.936392Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:d268dba8c36ed8cd0bfd5cd23086625b1710f3748f7a63a15b03b8a155dcc54d","observation_id":"be8c2661-0265-4886-bc12-eb28d06142bd","resolution":{"observed_at":"2026-08-07T12:51:33.625222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.430572Z","title":null,"venue":null,"work_id":"12c58a28-49f8-4d30-a61e-b12498be22ac","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.986989Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:ca1b205f646cb7a5741bf3c53628c9046dfdd1cea292448de3ede36c71b7c4d1","observation_id":"cb79c9a2-471d-45b0-82ac-8630a27a4e7a","resolution":{"observed_at":"2026-08-07T12:51:33.502194Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.272222Z","title":"primary-category","venue":null,"work_id":"e8dc7e19-bf38-4c3e-80bb-6e26156d7b68","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:32.078046Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:7ede7aada88ced841bc2f1a0b06a8be1ba695c103d567d88ddfb02fb64026944","observation_id":"cdb70ed6-3a38-4bdc-a165-88eb4604d332","resolution":{"observed_at":"2026-08-07T12:51:33.357836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.140351Z","title":null,"venue":null,"work_id":"25a00210-96ee-4743-885c-4cf8a77c69c1","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:32.165554Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:a84c42e6851b03a0952cfa733b9fcc812d0ce1e4cd9a9edb2f120f0841123998","observation_id":"9080c926-6ff6-4e9d-a2d2-ead3c7242945","resolution":{"observed_at":"2026-08-07T12:51:33.199731Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:33.049176Z","title":null,"venue":null,"work_id":"97c0e377-a2d2-4d57-84d9-5f6e0919b743","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:32.233456Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:83f1f137e16273b2f60f9931f7cb9a9bc7f91d93a2603793f7fa5afe735a2c09","observation_id":"6a80cdb7-7ab5-4335-822f-2ae8e5aa2e32","resolution":{"observed_at":"2026-08-07T12:51:33.083109Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:51:32.918080Z","title":null,"venue":null,"work_id":"8fbd80bc-de9c-4364-ad07-4ebbdf1b36df","year":null},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:32.342907Z"},"links":{"citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:ce0eb375d3b83033cef828721ad9b85b90af6781138099fd7a5caa21047e25bd","observation_id":"2987be69-2c7a-464a-a827-a700c925c63e","resolution":{"observed_at":"2026-08-07T12:51:33.000357Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19260","last_updated":"2025-01-02T18:46:05Z","snapshot_observed_at":"2026-07-06T20:13:27.068339Z","submitted_at":"2024-12-26T15:54:10Z","title":"MEDEC: A Benchmark for Medical Error Detection and Correction in Clinical Notes","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19260","snapshot_observed_at":"2026-08-07T12:51:31.281190Z","title":"whear”, “histori","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T12:51:31.281190Z"},"links":{"cited_paper":"/paper/2412.19260","citing_paper":"/paper/2506.00064"},"observation_digest":"sha256:85ef5747fd787aaab922d757de8a9bb3d77ad5b3a4c7fd177e5997c667c42110","observation_id":"472a067e-06df-4aa6-b763-7a6f01bde976","resolution":{"observed_at":"2026-08-07T12:51:31.281190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.00064","last_updated":"2025-05-29T13:52:58Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T12:43:16.451780Z","submitted_at":"2025-05-29T13:52:58Z","title":"Mis-prompt: Benchmarking Large Language Models for Proactive Error Handling"},"reference_resolution":{"displayed":15,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":0,"verified_fuzzy":6},"total_outbound_references":15},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 15 of 15 outbound references and 0 inbound Pith citation observations for arXiv:2506.00064."}