{"as_of":"2026-08-09T02:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b4c967ffd89754deedcbef3cbac3379931153b89267a3d11028dffb050e3f950","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:25:00.068070Z","state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-26T00:50:22.839005Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T16:09:57.586338Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"cited_work":{"arxiv_id":"2505.15282","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15282","snapshot_observed_at":"2026-07-04T16:09:57.586338Z","title":"arXiv preprint arXiv:2505.15282 (2025) 4, 10, 22, 24","venue":null,"work_id":"cde93db2-817c-405f-a2c1-e6db3dbf45bb","year":2025},"citing_paper":{"arxiv_id":"2606.24333","last_updated":"2026-06-23T09:11:24Z","snapshot_observed_at":"2026-07-06T23:58:54.018557Z","submitted_at":"2026-06-23T09:11:24Z","title":"UniTranslator: A Unified Multi-modal Framework for End-to-end In-Image Machine Translation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-26T00:50:22.839005Z"},"links":{"cited_paper":"/paper/2505.15282","citing_paper":"/paper/2606.24333"},"observation_digest":"sha256:a0e856756cc881e77dc2dca0451dac40b79223c8b9bdbda191dee5f9db6765a2","observation_id":"2c5d54f7-2fc9-405d-92e1-1d21404be97d","resolution":{"observed_at":"2026-07-04T16:09:57.587855Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.15282/citation-record","integrity":"/paper/2505.15282/integrity","json":"/paper/2505.15282/citation-record.json","paper":"/paper/2505.15282"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2202.04200","last_updated":"2022-02-08T23:54:06Z","snapshot_observed_at":"2026-08-06T14:19:25.667074Z","submitted_at":"2022-02-08T23:54:06Z","title":"MaskGIT: Masked Generative Image Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.04200","snapshot_observed_at":"2026-08-07T15:24:58.017560Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.017560Z"},"links":{"cited_paper":"/paper/2202.04200","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:f7652af1e252d48c780e01c7df4b864d762d0f57fcfe70803351358059bcdc5c","observation_id":"6709c8a0-b713-4307-bfc3-516aa7d591b1","resolution":{"observed_at":"2026-08-07T15:24:58.017560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-07T15:24:58.083476Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.083476Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:9ffb0214e1e756dab984d7f0ac5a4f9b5ebdb321aa83d48866dbf51bfc36ab1e","observation_id":"bff4345f-620d-49ba-afa7-d9d175224f55","resolution":{"observed_at":"2026-08-07T15:24:58.083476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:24:58.191247Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.191247Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:9fe0fafb58db4203890abc8042ed7aec559b6a584c4a741aa4cb6a681f8b5719","observation_id":"b5594387-5a53-4a3e-8241-afe73be37fdc","resolution":{"observed_at":"2026-08-07T15:24:58.191247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:24:58.285210Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.285210Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:7507a71c38f62fe4172cea0aec26426d7d5e6f95953fad200b80a6af16260281","observation_id":"440ee06c-54e0-421a-8c29-fa0163688338","resolution":{"observed_at":"2026-08-07T15:24:58.285210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1706.08500","last_updated":"2018-01-12T14:05:44Z","snapshot_observed_at":"2026-07-06T05:48:30.254634Z","submitted_at":"2017-06-26T17:45:23Z","title":"GANs Trained by a Two Time-Scale Update Rule Converge to a Local Nash Equilibrium","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.08500","snapshot_observed_at":"2026-08-07T15:24:58.360512Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.360512Z"},"links":{"cited_paper":"/paper/1706.08500","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:10c265a95e734f9ba6be1b88fb713488a581df4599acf5e740b4bc412337047c","observation_id":"7ae6ea06-7870-47fa-8be7-1357fa89f99b","resolution":{"observed_at":"2026-08-07T15:24:58.360512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:24:58.423561Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.423561Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:ca4738156cf8b093fe68f21c06b545f6d49471bf200f723e0b3a8db837655348","observation_id":"3e0a514b-2598-49b0-aa7f-8d584fd6949c","resolution":{"observed_at":"2026-08-07T15:24:58.423561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.02894","last_updated":"2024-07-03T08:15:39Z","snapshot_observed_at":"2026-07-06T18:40:48.783743Z","submitted_at":"2024-07-03T08:15:39Z","title":"Translatotron-V(ison): An End-to-End Model for In-Image Machine Translation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.02894","snapshot_observed_at":"2026-08-07T15:24:58.507827Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.507827Z"},"links":{"cited_paper":"/paper/2407.02894","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:99a1536459006710defe9f7895bc78adac632fd29c9c652ec1fa9cbc466724fc","observation_id":"db4b5294-ab29-4214-bc29-4b639d7ff6a7","resolution":{"observed_at":"2026-08-07T15:24:58.507827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17415","last_updated":"2023-06-02T12:38:37Z","snapshot_observed_at":"2026-07-06T15:34:16.800916Z","submitted_at":"2023-05-27T08:41:18Z","title":"Exploring Better Text Image Translation with Multimodal Codebook","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.17415","snapshot_observed_at":"2026-08-07T15:24:58.596808Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.596808Z"},"links":{"cited_paper":"/paper/2305.17415","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:28d0e69f7ca08bb556d318db38fcb6edf1cbdf92030a59a13ac4469f38f1c680","observation_id":"72e5d6a0-198c-4fa6-9e19-27f0715303af","resolution":{"observed_at":"2026-08-07T15:24:58.596808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-07T15:24:58.716178Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.716178Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:cc8b3abdf4436437b27a911a66272fa22200025c85c635873c2614777e737906","observation_id":"012a442d-c509-45c9-9bc8-bd410d5f23f9","resolution":{"observed_at":"2026-08-07T15:24:58.716178Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.10337","last_updated":"2018-10-29T15:34:15Z","snapshot_observed_at":"2026-07-06T06:11:35.990430Z","submitted_at":"2017-11-28T15:19:53Z","title":"Are GANs Created Equal? A Large-Scale Study","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.10337","snapshot_observed_at":"2026-08-07T15:24:58.772647Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.772647Z"},"links":{"cited_paper":"/paper/1711.10337","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:78a4ed14b19f230e38d23663674b3b852c36f774246cf58e53a2cebfa1a5f29c","observation_id":"562d297c-4531-44da-976c-f068fdbec7d1","resolution":{"observed_at":"2026-08-07T15:24:58.772647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03887","last_updated":"2022-10-08T02:35:45Z","snapshot_observed_at":"2026-08-09T00:53:52.087368Z","submitted_at":"2022-10-08T02:35:45Z","title":"Improving End-to-End Text Image Translation From the Auxiliary Text Translation Task","version":1},"cited_work":{"arxiv_id":"2210.03887","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.03887","snapshot_observed_at":"2026-08-07T15:25:00.597835Z","title":"Improving End-to-End Text Image Translation From the Auxiliary Text Translation Task","venue":"cs.CL","work_id":"f6cf3065-1ffc-4ba7-ab35-70793eb2987c","year":2022},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.857980Z"},"links":{"cited_paper":"/paper/2210.03887","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:9247e62555a622d60ed0a5cf52201147a4cd69cb1e7974893d756ae8f7e34a19","observation_id":"ff725a69-e941-4c12-959e-ae56e3005f23","resolution":{"observed_at":"2026-08-07T15:25:00.686807Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.findings-emnlp.330","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:25:00.171972Z","title":null,"venue":null,"work_id":"3ecaf148-4ae3-418d-b77c-7eb488d5a9d5","year":2023},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.905517Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:d5ee3481f2c40cb187f8d83b1046c69301a988072c8fbd607ec9c08dac43696d","observation_id":"f231c192-59aa-4f83-941e-839862823395","resolution":{"observed_at":"2026-08-07T15:25:00.177956Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17870","last_updated":"2023-05-23T04:07:00Z","snapshot_observed_at":"2026-08-07T22:11:51.398507Z","submitted_at":"2023-03-31T08:06:33Z","title":"GlyphDraw: Seamlessly Rendering Text with Intricate Spatial Structures in Text-to-Image Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17870","snapshot_observed_at":"2026-08-07T15:24:58.976929Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:58.976929Z"},"links":{"cited_paper":"/paper/2303.17870","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:d2328a0b15a092d0c6baa431943a6c197d38520f25cddc4e1321c15fb06a1aa1","observation_id":"53acded9-25d9-4c3e-b083-de736ed51637","resolution":{"observed_at":"2026-08-07T15:24:58.976929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:24:59.013450Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:59.013450Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:3be6e6cd08158d9797910bdcefb4e28e1ec5d48c89975acd430acaa181828796","observation_id":"314f1857-49ad-4e20-b626-9d056dfcf22e","resolution":{"observed_at":"2026-08-07T15:24:59.013450Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:24:59.120479Z","title":null,"venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:59.120479Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:e7da3097dd0d985debdc8c72f5f12d3fe612ba8b09d0a641bcaf575541a0c0af","observation_id":"285303f1-240a-40c8-8637-cb990d9311dc","resolution":{"observed_at":"2026-08-07T15:24:59.120479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:24:59.187630Z","title":"Wong, Xiaoshuai Sun, and Rongrong Ji","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:59.187630Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:e3eb78dff46a3fe9de87727e5494157f14ab9f2caf58c20449a1793f51504b10","observation_id":"cd3e0f54-bfa2-4567-ab93-5090d1981fc1","resolution":{"observed_at":"2026-08-07T15:24:59.187630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:24:59.239491Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:59.239491Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:fd91624f480fe00e23aa40196e2221599eaa812650f6b60a87d80c695a0f2921","observation_id":"d69259e4-94b4-412b-a9ee-0d6c18ea153c","resolution":{"observed_at":"2026-08-07T15:24:59.239491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:25:00.930206Z","title":"Rodr \\' guez, David Vazquez, Issam Laradji, Marco Pedersoli, and Pau Rodriguez","venue":null,"work_id":"8d7902f2-76be-4b3e-95d5-14fbfd9ea6b0","year":2023},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:59.336431Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:0c8e1ab7086ee064946c4c52a53034734d3f401c8ea1071b85c7ec9bf4aa8239","observation_id":"87ade139-aa16-411b-968e-69d270d1431c","resolution":{"observed_at":"2026-08-07T15:25:01.073016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:24:59.406120Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:59.406120Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:813ba42abd72bb06bf8559fc15b345e7aad51086dba48734866d8e877fb7bfd5","observation_id":"e0e66f42-5407-4386-8af2-4ec7afa1fc5d","resolution":{"observed_at":"2026-08-07T15:24:59.406120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02905","last_updated":"2024-06-10T17:59:07Z","snapshot_observed_at":"2026-07-06T17:55:16.774337Z","submitted_at":"2024-04-03T17:59:53Z","title":"Visual Autoregressive Modeling: Scalable Image Generation via Next-Scale Prediction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02905","snapshot_observed_at":"2026-08-07T15:24:59.516626Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:59.516626Z"},"links":{"cited_paper":"/paper/2404.02905","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:e9511bce842c0153fa7f8b3a512ed7c8b649fcc777a9badfb0ce933fe95d1101","observation_id":"956fea12-608a-4243-b21f-f75bd6803d44","resolution":{"observed_at":"2026-08-07T15:24:59.516626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:24:59.577304Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:59.577304Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:87166f4e0116c8004ef57710dd8135f39aa140dc1d67ce856cc5d813dac80c17","observation_id":"789ee9d1-7d2b-49a8-a7f2-a96763a98179","resolution":{"observed_at":"2026-08-07T15:24:59.577304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.03054","last_updated":"2024-02-21T07:12:45Z","snapshot_observed_at":"2026-08-07T22:13:54.617491Z","submitted_at":"2023-11-06T12:10:43Z","title":"AnyText: Multilingual Visual Text Generation And Editing","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.03054","snapshot_observed_at":"2026-08-07T15:24:59.675792Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:59.675792Z"},"links":{"cited_paper":"/paper/2311.03054","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:5be22946ff02179b9ed737244aa0f46254487a4ce83ca4597afcb6b813900a49","observation_id":"b88939f1-dbb8-4b0b-95d0-e48553bc49e8","resolution":{"observed_at":"2026-08-07T15:24:59.675792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1706.03762","last_updated":"2023-08-02T00:41:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-06-12T17:57:34Z","title":"Attention Is All You Need","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.03762","snapshot_observed_at":"2026-08-07T15:24:59.806307Z","title":"Gomez, Lukasz Kaiser, and Illia Polosukhin","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:59.806307Z"},"links":{"cited_paper":"/paper/1706.03762","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:92ef35f4ce183a36e543cfa54d9c45927942d14dd81751ae7c64f0bf1c5da166","observation_id":"22a951cf-aba0-4d5a-a1ed-eb97af525dc8","resolution":{"observed_at":"2026-08-07T15:24:59.806307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:24:59.961023Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T15:24:59.961023Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:505f1729bb40157f1a666fbbd869a8f8e4fd951b07eb91e14e5b5fdd000b9337","observation_id":"be185552-b67d-4362-896d-d01dab7c1397","resolution":{"observed_at":"2026-08-07T15:24:59.961023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12232","last_updated":"2023-12-19T15:18:40Z","snapshot_observed_at":"2026-07-06T17:05:19.007669Z","submitted_at":"2023-12-19T15:18:40Z","title":"Brush Your Text: Synthesize Any Scene Text on Images via Diffusion Model","version":1},"cited_work":{"arxiv_id":"2312.12232","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.12232","snapshot_observed_at":"2026-08-07T15:25:00.363329Z","title":"Brush Your Text: Synthesize Any Scene Text on Images via Diffusion Model","venue":"cs.CV","work_id":"4303ae5c-c480-4e83-9a48-4ae4b95584c8","year":2023},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T15:25:00.042892Z"},"links":{"cited_paper":"/paper/2312.12232","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:342406b3929759e8e56e3ad181b82f5f2a6c36e26ba9ea5f8bf64e0b9c8d8e95","observation_id":"bc0411b3-c2a7-43d2-ac83-80ad18e5b853","resolution":{"observed_at":"2026-08-07T15:25:00.463920Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1801.03924","last_updated":"2018-04-10T19:25:07Z","snapshot_observed_at":"2026-08-08T03:05:32.889595Z","submitted_at":"2018-01-11T18:54:17Z","title":"The Unreasonable Effectiveness of Deep Features as a Perceptual Metric","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1801.03924","snapshot_observed_at":"2026-08-07T15:25:00.047785Z","title":"Efros, Eli Shechtman, and Oliver Wang","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T15:25:00.047785Z"},"links":{"cited_paper":"/paper/1801.03924","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:946feed9e4b1ab64548fb671ce3ddef97ca7c64f9eb8e4f67835342471eff7ec","observation_id":"c391a9d8-ee3e-4172-a87a-51be8753fd17","resolution":{"observed_at":"2026-08-07T15:25:00.047785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:25:00.052667Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T15:25:00.052667Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:dc392a200c129d51f9975ffac5702f960c5a4129194fef98bb74d55f1c85a119","observation_id":"53772ca3-66fb-4321-a1a3-77123e7f289f","resolution":{"observed_at":"2026-08-07T15:25:00.052667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.08520","last_updated":"2020-05-18T08:23:41Z","snapshot_observed_at":"2026-08-08T19:46:40.577350Z","submitted_at":"2020-05-18T08:23:41Z","title":"Robust Training of Vector Quantized Bottleneck Models","version":1},"cited_work":{"arxiv_id":"2005.08520","doi":null,"metadata_source":"pith","pith_arxiv_id":"2005.08520","snapshot_observed_at":"2026-08-07T15:25:00.220614Z","title":"Robust Training of Vector Quantized Bottleneck Models","venue":"cs.LG","work_id":"de29d3a5-9606-46dc-a46c-a66ec49db986","year":2020},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T15:25:00.057539Z"},"links":{"cited_paper":"/paper/2005.08520","citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:5a0fad8bcc376431b62ac551ccb799c9a5100ee503be3973a0ec26eddb32ad92","observation_id":"89ec5e2f-7deb-4529-81f8-df349014077a","resolution":{"observed_at":"2026-08-07T15:25:00.259566Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:25:00.062618Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T15:25:00.062618Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:ae0f4395e74daa8ab0e34e0e0161571ced4fb45970734317da1df56568818872","observation_id":"7c7cd0df-d723-4fd6-840e-88c46227316d","resolution":{"observed_at":"2026-08-07T15:25:00.062618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:25:00.068070Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T15:25:00.068070Z"},"links":{"citing_paper":"/paper/2505.15282"},"observation_digest":"sha256:da600e106027cfc5d2e5b92d8d54e1cc12780839ed75cb38d67f8b6be2c1f48e","observation_id":"595342a6-bfe3-4903-9f6b-33513cd352cb","resolution":{"observed_at":"2026-08-07T15:25:00.068070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.15282","last_updated":"2025-05-21T09:02:53Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T22:12:37.154313Z","submitted_at":"2025-05-21T09:02:53Z","title":"Exploring In-Image Machine Translation with Real-World Background"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":25,"verified_exact":4,"verified_fuzzy":1},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 1 inbound Pith citation observation for arXiv:2505.15282."}