{"as_of":"2026-08-15T13:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8f215a75a37d4bf5b1e297102ad67c2c6129bb137d2817712f64f495bf1c4a2d","coverage":[{"denominator":19,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T11:24:16.101773Z","state":"measured"},{"denominator":19,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":19,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.15523/citation-record","integrity":"/paper/2412.15523/integrity","json":"/paper/2412.15523/citation-record.json","paper":"/paper/2412.15523"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2307.12270","last_updated":"2023-10-09T05:48:11Z","snapshot_observed_at":"2026-08-14T02:56:55.701205Z","submitted_at":"2023-07-23T09:04:13Z","title":"Context Perception Parallel Decoder for Scene Text Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.12270","snapshot_observed_at":"2026-08-11T11:24:16.029155Z","title":"arXiv preprint arXiv:2307.12270","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.029155Z"},"links":{"cited_paper":"/paper/2307.12270","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:48ec2a13073de9b24468ff2ee45a486396227f6ec1213c21c97c71e7878ffce1","observation_id":"9ed2e4b1-4105-4225-b75c-1c21a627f890","resolution":{"observed_at":"2026-08-11T11:24:16.029155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.03895","last_updated":"2023-09-07T17:56:57Z","snapshot_observed_at":"2026-08-13T19:26:36.417242Z","submitted_at":"2023-09-07T17:56:57Z","title":"InstructDiffusion: A Generalist Modeling Interface for Vision Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.03895","snapshot_observed_at":"2026-08-11T11:24:16.033705Z","title":"arXiv preprint arXiv:2309.03895","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.033705Z"},"links":{"cited_paper":"/paper/2309.03895","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:dcd4028ca235503d81f8fb8f743372108e184897a4b4b7784857d34255e28c51","observation_id":"67e2161a-9d07-4967-993d-74c303dda212","resolution":{"observed_at":"2026-08-11T11:24:16.033705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.15664","last_updated":"2022-10-06T06:50:39Z","snapshot_observed_at":"2026-08-13T17:24:28.063798Z","submitted_at":"2021-11-30T18:55:19Z","title":"OCR-free Document Understanding Transformer","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.15664","snapshot_observed_at":"2026-08-11T11:24:16.048960Z","title":"arXiv preprint arXiv:2111.15664, 7:","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.048960Z"},"links":{"cited_paper":"/paper/2111.15664","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:4aefc70bfddb8dfb167f017d67df6f56808f60057fbde31b673b05e8b0d4e0e1","observation_id":"7cbb866a-e573-4159-aeec-6e46a2ad7c9a","resolution":{"observed_at":"2026-08-11T11:24:16.048960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.02643","last_updated":"2023-04-05T17:59:46Z","snapshot_observed_at":"2026-08-08T05:14:59.435033Z","submitted_at":"2023-04-05T17:59:46Z","title":"Segment Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.02643","snapshot_observed_at":"2026-08-11T11:24:16.054357Z","title":"arXiv preprint arXiv:2304.02643","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.054357Z"},"links":{"cited_paper":"/paper/2304.02643","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:4f1d4bf772bb3999463453709e64ee1811dd3579502c8b7a52da785ec54cd035","observation_id":"7c1003f8-8de2-47c5-8774-3b8ac79bff0f","resolution":{"observed_at":"2026-08-11T11:24:16.054357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-11T11:24:16.059498Z","title":"arXiv preprint arXiv:2303.05499","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.059498Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:a73fbf6c927f832e2cc41b23647d9d17da84350654a3f908ae042fe7e3e7f06d","observation_id":"55638be9-ac1e-4fa9-b141-dde60e63ca80","resolution":{"observed_at":"2026-08-11T11:24:16.059498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.01635","last_updated":"2023-09-02T05:01:23Z","snapshot_observed_at":"2026-08-14T18:48:54.629353Z","submitted_at":"2023-01-04T14:20:14Z","title":"SPTS v2: Single-Point Scene Text Spotting","version":4},"cited_work":{"arxiv_id":"2301.01635","doi":null,"metadata_source":"pith","pith_arxiv_id":"2301.01635","snapshot_observed_at":"2026-08-11T11:24:16.216968Z","title":"SPTS v2: Single-Point Scene Text Spotting","venue":"cs.CV","work_id":"92dca76c-5a26-46f9-9ff1-7d8f31fa9e3f","year":2023},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.064952Z"},"links":{"cited_paper":"/paper/2301.01635","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:814ab355c43b92f2734560a7e2834d08cbedd4fa4d70e183c71845a9be588945","observation_id":"bcfc6fd0-f36c-4591-a414-ec1fbebe0002","resolution":{"observed_at":"2026-08-11T11:24:16.222183Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.11419","last_updated":"2024-08-21T16:54:23Z","snapshot_observed_at":"2026-08-13T10:09:07.558559Z","submitted_at":"2023-09-20T15:50:08Z","title":"KOSMOS-2.5: A Multimodal Literate Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.11419","snapshot_observed_at":"2026-08-11T11:24:16.070213Z","title":"arXiv preprint arXiv:2309.11419","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.070213Z"},"links":{"cited_paper":"/paper/2309.11419","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:8d9254724b29269457bf5c57eacc3fe7ed26798848b4c589bc32db4a90c1c647","observation_id":"ee1b0aa8-656e-4374-b20a-32bffae4b191","resolution":{"observed_at":"2026-08-11T11:24:16.070213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T11:24:16.359809Z","title":"In 2017 14th IAPR international con- ference on document analysis and recognition (ICDAR), vol- ume 1, 1454–1459","venue":null,"work_id":"bb7e0c67-8e13-4996-ae7a-e6546e84e77d","year":2017},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.075726Z"},"links":{"citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:27c574006d761061aa8b1e013b64a9607e96d6c76d27a2eed75f450202d85f0f","observation_id":"08b6e3a0-caf1-43b4-82ba-31959c3f891f","resolution":{"observed_at":"2026-08-11T11:24:16.365884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02694","last_updated":"2026-05-26T02:17:17Z","snapshot_observed_at":"2026-08-13T05:09:37.913989Z","submitted_at":"2023-12-05T11:53:17Z","title":"UPOCR: Towards Unified Pixel-Level OCR Interface","version":2},"cited_work":{"arxiv_id":"2312.02694","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.02694","snapshot_observed_at":"2026-08-11T11:24:16.178345Z","title":"UPOCR: Towards Unified Pixel-Level OCR Interface","venue":"cs.CV","work_id":"b710c71e-f254-4d90-89e7-9da0a5fc5d78","year":2023},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.088303Z"},"links":{"cited_paper":"/paper/2312.02694","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:3515d35ef82d1ed9987c2be99d316dacd573fc0e6113939b88d192642ed45c7b","observation_id":"4dac2de8-9d73-4393-ae63-54579e5874c6","resolution":{"observed_at":"2026-08-11T11:24:16.185327Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05126","last_updated":"2023-10-08T11:33:09Z","snapshot_observed_at":"2026-08-13T05:54:33.801989Z","submitted_at":"2023-10-08T11:33:09Z","title":"UReader: Universal OCR-free Visually-situated Language Understanding with Multimodal Large Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05126","snapshot_observed_at":"2026-08-11T11:24:16.101773Z","title":"arXiv preprint arXiv:2310.05126","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.101773Z"},"links":{"cited_paper":"/paper/2310.05126","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:b2dd33a4b2dfa0e61c367a750cb7adaf8aec4041a7f60b3f58b3dd41db31816e","observation_id":"c6d0c6cb-54eb-499f-adce-c11763938a7a","resolution":{"observed_at":"2026-08-11T11:24:16.101773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T11:24:16.391928Z","title":"In 12th international conference on document analysis and recognition, 1484–1493","venue":null,"work_id":"cf46360c-8bab-4360-a596-fcd86c0ea62b","year":2013},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2013,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.039342Z"},"links":{"citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:1023367d72ea6a7904930ff5ad9c5a389e197f6009f75b02b73115c3155133d8","observation_id":"f113a607-6fe4-4838-bc18-46fc8ff6dbc9","resolution":{"observed_at":"2026-08-11T11:24:16.397469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T11:24:16.376344Z","title":"In 13th international conference on document analysis and recognition, 1156–1160","venue":null,"work_id":"bc966fa6-a17e-413a-8064-acdedbe6a08f","year":2015},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.044359Z"},"links":{"citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:af41587851347b0315cf6b7ef05274e296185f018de7fa4653d4468517b3d204","observation_id":"8f093086-639d-4626-b5c1-3082f28b7b1b","resolution":{"observed_at":"2026-08-11T11:24:16.381022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T11:24:16.407785Z","title":"In 2017 14th IAPR international conference on document anal- ysis and recognition (ICDAR), volume 1, 935–942","venue":null,"work_id":"c9bb1029-7746-436c-87a8-4532c22623a9","year":2017},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.013602Z"},"links":{"citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:336cf74f0d9420c55c287bec32767a46d1ad114058567a733a6cb41ea84bc9de","observation_id":"02c21729-2244-4df4-b7dc-09dfb6d184a5","resolution":{"observed_at":"2026-08-11T11:24:16.413833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-08-14T18:16:28.847993Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-11T11:24:16.017859Z","title":"arXiv preprint arXiv:1810.04805","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.017859Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:50c33929b0a910798f13ef6f6d3f72fbb44bd328d31ec0ddf0a1f348afcd0cc1","observation_id":"d915012a-b1e2-44be-92fa-3a29e0daee91","resolution":{"observed_at":"2026-08-11T11:24:16.017859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T14:19:26.598265Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-11T11:24:16.024051Z","title":"arXiv preprint arXiv:2010.11929","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.024051Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:86b3332ee643fa07647746103f25dad950c90e0b0cfca75b19b2c72404d4ffec","observation_id":"a5d838e3-9ea4-449b-9aac-ab771b998daa","resolution":{"observed_at":"2026-08-11T11:24:16.024051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.10852","last_updated":"2022-03-27T14:44:00Z","snapshot_observed_at":"2026-08-14T14:29:18.640120Z","submitted_at":"2021-09-22T17:26:36Z","title":"Pix2seq: A Language Modeling Framework for Object Detection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.10852","snapshot_observed_at":"2026-08-11T11:24:16.009164Z","title":"arXiv preprint arXiv:2109.10852","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.009164Z"},"links":{"cited_paper":"/paper/2109.10852","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:412b6634d46e1c135de60de861d6079f184fbb1cdb193253415ace9b0b4484d6","observation_id":"3dca4bb9-7760-4cb3-8a55-75c53a62262a","resolution":{"observed_at":"2026-08-11T11:24:16.009164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.14100","last_updated":"2022-12-15T19:21:35Z","snapshot_observed_at":"2026-08-04T09:31:34.784365Z","submitted_at":"2022-05-27T17:03:38Z","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.14100","snapshot_observed_at":"2026-08-11T11:24:16.096991Z","title":"arXiv preprint arXiv:2205.14100","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.096991Z"},"links":{"cited_paper":"/paper/2205.14100","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:c8cb1b4782fb0564eaf3ac1c613da4e3d535a14060923dbaa343e3407f9e5f5f","observation_id":"7507f045-ef53-4b51-9d07-75b3bad64f5a","resolution":{"observed_at":"2026-08-11T11:24:16.096991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-11T11:24:16.002128Z","title":"arXiv preprint arXiv:2308.12966","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.002128Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:96a4091c4731656aefad457c4c7eec70b81b0b9c4843e62b96c412fd197ae711","observation_id":"c522d627-393f-42a0-9000-e8c0a3b9b3d5","resolution":{"observed_at":"2026-08-11T11:24:16.002128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19128","last_updated":"2024-03-28T03:51:14Z","snapshot_observed_at":"2026-08-14T02:30:03.077140Z","submitted_at":"2024-03-28T03:51:14Z","title":"OmniParser: A Unified Framework for Text Spotting, Key Information Extraction and Table Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19128","snapshot_observed_at":"2026-08-11T11:24:16.092992Z","title":"arXiv preprint arXiv:2403.19128","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.092992Z"},"links":{"cited_paper":"/paper/2403.19128","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:9aec877b9130db15feee2b8dbed880024f9e33c9775028182bd8df8f5dd30fe0","observation_id":"8c07e379-aa0b-41b6-927f-a3aafe304c2c","resolution":{"observed_at":"2026-08-11T11:24:16.092992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T22:23:08.072442Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting"},"reference_resolution":{"displayed":19,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":13,"verified_exact":0,"verified_fuzzy":4},"total_outbound_references":19},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 19 of 19 outbound references and 0 inbound Pith citation observations for arXiv:2412.15523."}