{"as_of":"2026-08-23T20:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5e1651056f5b6a66b2c586002b39b965904e505d4878d10bb9cc4f5b824adc5f","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":7,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":7,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-10T18:58:32.854964Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T19:07:35.356549Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.06312","last_updated":"2025-07-22T15:05:39Z","snapshot_observed_at":"2026-08-18T20:11:07.112891Z","submitted_at":"2025-03-08T19:10:04Z","title":"DOFA-CLIP: Multimodal Vision-Language Foundation Models for Earth Observation","version":2},"cited_work":{"arxiv_id":"2503.06312","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.06312","snapshot_observed_at":"2026-07-10T19:07:35.356549Z","title":"Dofa-clip: Multimodal vision-language foundation models for earth observation","venue":"cs.CV","work_id":"7f42dcae-6d9d-401f-b5dc-f6cfe17a20a8","year":2025},"citing_paper":{"arxiv_id":"2604.24919","last_updated":"2026-06-01T00:50:05Z","snapshot_observed_at":"2026-08-15T03:11:06.203859Z","submitted_at":"2026-04-27T18:59:49Z","title":"Agentic AI for Remote Sensing: Technical Challenges and Research Directions","version":1},"reference_index":129,"source":"pdf_text","source_observed_at":"2026-05-08T04:29:22.477531Z"},"links":{"cited_paper":"/paper/2503.06312","citing_paper":"/paper/2604.24919"},"observation_digest":"sha256:76116dc1addf60d397ece04a1106c14260ccdaf36b38e125eee72947daad9e62","observation_id":"591dff69-1037-49f5-92ff-d476c0bd1884","resolution":{"observed_at":"2026-05-11T21:46:28.598350Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06312","last_updated":"2025-07-22T15:05:39Z","snapshot_observed_at":"2026-08-18T20:11:07.112891Z","submitted_at":"2025-03-08T19:10:04Z","title":"DOFA-CLIP: Multimodal Vision-Language Foundation Models for Earth Observation","version":2},"cited_work":{"arxiv_id":"2503.06312","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.06312","snapshot_observed_at":"2026-07-10T19:07:35.356549Z","title":"Dofa-clip: Multimodal vision-language foundation models for earth observation","venue":"cs.CV","work_id":"7f42dcae-6d9d-401f-b5dc-f6cfe17a20a8","year":2025},"citing_paper":{"arxiv_id":"2604.24919","last_updated":"2026-06-01T00:50:05Z","snapshot_observed_at":"2026-08-15T03:11:06.203859Z","submitted_at":"2026-04-27T18:59:49Z","title":"Agentic AI for Remote Sensing: Technical Challenges and Research Directions","version":2},"reference_index":129,"source":"pdf_text","source_observed_at":"2026-05-14T20:55:38.841743Z"},"links":{"cited_paper":"/paper/2503.06312","citing_paper":"/paper/2604.24919"},"observation_digest":"sha256:95b899e842c75b4f77783d11cafacc6bfdb80d4021e6741a0fa0773a42ee6f45","observation_id":"02d7544c-640c-4f46-8cfe-63a54f3a3527","resolution":{"observed_at":"2026-05-14T20:59:27.708342Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06312","last_updated":"2025-07-22T15:05:39Z","snapshot_observed_at":"2026-08-18T20:11:07.112891Z","submitted_at":"2025-03-08T19:10:04Z","title":"DOFA-CLIP: Multimodal Vision-Language Foundation Models for Earth Observation","version":2},"cited_work":{"arxiv_id":"2503.06312","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.06312","snapshot_observed_at":"2026-07-10T19:07:35.356549Z","title":"Dofa-clip: Multimodal vision-language foundation models for earth observation","venue":"cs.CV","work_id":"7f42dcae-6d9d-401f-b5dc-f6cfe17a20a8","year":2025},"citing_paper":{"arxiv_id":"2605.06990","last_updated":"2026-05-07T22:10:41Z","snapshot_observed_at":"2026-08-11T14:45:51.625191Z","submitted_at":"2026-05-07T22:10:41Z","title":"TRAJGANR: Trajectory-Centric Urban Multimodal Learning via Geospatially Aligned Neural Representations","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-11T01:27:50.566355Z"},"links":{"cited_paper":"/paper/2503.06312","citing_paper":"/paper/2605.06990"},"observation_digest":"sha256:5b43ffab38e239a81d8f47b726b93cc29692a67f0dcb9aa455dfb0d1855cb102","observation_id":"6a7fcc82-a531-48e9-870d-778af0dd1f36","resolution":{"observed_at":"2026-05-11T04:20:59.395954Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06312","last_updated":"2025-07-22T15:05:39Z","snapshot_observed_at":"2026-08-18T20:11:07.112891Z","submitted_at":"2025-03-08T19:10:04Z","title":"DOFA-CLIP: Multimodal Vision-Language Foundation Models for Earth Observation","version":2},"cited_work":{"arxiv_id":"2503.06312","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.06312","snapshot_observed_at":"2026-07-10T19:07:35.356549Z","title":"Dofa-clip: Multimodal vision-language foundation models for earth observation","venue":"cs.CV","work_id":"7f42dcae-6d9d-401f-b5dc-f6cfe17a20a8","year":2025},"citing_paper":{"arxiv_id":"2605.12542","last_updated":"2026-06-11T05:51:05Z","snapshot_observed_at":"2026-08-13T21:10:05.886370Z","submitted_at":"2026-05-09T08:34:30Z","title":"Earth Science Foundation Models: From Perception to Reasoning and Discovery","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-14T22:07:40.242567Z"},"links":{"cited_paper":"/paper/2503.06312","citing_paper":"/paper/2605.12542"},"observation_digest":"sha256:bdde2be83894c2e8b4add49d082403fc436e681b466faef2eb0c4776b2bd965f","observation_id":"827d56f2-fcb4-423d-bbd9-ea1005193987","resolution":{"observed_at":"2026-05-14T22:08:03.674103Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06312","last_updated":"2025-07-22T15:05:39Z","snapshot_observed_at":"2026-08-18T20:11:07.112891Z","submitted_at":"2025-03-08T19:10:04Z","title":"DOFA-CLIP: Multimodal Vision-Language Foundation Models for Earth Observation","version":2},"cited_work":{"arxiv_id":"2503.06312","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.06312","snapshot_observed_at":"2026-07-10T19:07:35.356549Z","title":"Dofa-clip: Multimodal vision-language foundation models for earth observation","venue":"cs.CV","work_id":"7f42dcae-6d9d-401f-b5dc-f6cfe17a20a8","year":2025},"citing_paper":{"arxiv_id":"2605.12542","last_updated":"2026-06-11T05:51:05Z","snapshot_observed_at":"2026-08-13T21:10:05.886370Z","submitted_at":"2026-05-09T08:34:30Z","title":"Earth Science Foundation Models: From Perception to Reasoning and Discovery","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-30T23:07:21.558834Z"},"links":{"cited_paper":"/paper/2503.06312","citing_paper":"/paper/2605.12542"},"observation_digest":"sha256:58e3a311fd009732aac8561914f154904ab4f55688d25d1eed0d398c559dd11c","observation_id":"365b50f6-2c49-4b7b-806f-5832998b6a4a","resolution":{"observed_at":"2026-07-01T13:35:46.220132Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06312","last_updated":"2025-07-22T15:05:39Z","snapshot_observed_at":"2026-08-18T20:11:07.112891Z","submitted_at":"2025-03-08T19:10:04Z","title":"DOFA-CLIP: Multimodal Vision-Language Foundation Models for Earth Observation","version":2},"cited_work":{"arxiv_id":"2503.06312","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.06312","snapshot_observed_at":"2026-07-10T19:07:35.356549Z","title":"Dofa-clip: Multimodal vision-language foundation models for earth observation","venue":"cs.CV","work_id":"7f42dcae-6d9d-401f-b5dc-f6cfe17a20a8","year":2025},"citing_paper":{"arxiv_id":"2605.15397","last_updated":"2026-05-14T20:30:25Z","snapshot_observed_at":"2026-08-03T23:32:42.112407Z","submitted_at":"2026-05-14T20:30:25Z","title":"ELDOR: A Dataset and Benchmark for Illegal Gold Mining in the Amazon Rainforest","version":1},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-05-19T15:35:25.103868Z"},"links":{"cited_paper":"/paper/2503.06312","citing_paper":"/paper/2605.15397"},"observation_digest":"sha256:1449809dc764fbbcab0d2d8283a705494fa91425f0e02502b10b3eb6b90b762f","observation_id":"e5c441da-20e8-4e0f-a391-79e8dbf5b186","resolution":{"observed_at":"2026-05-19T15:37:37.241551Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06312","last_updated":"2025-07-22T15:05:39Z","snapshot_observed_at":"2026-08-18T20:11:07.112891Z","submitted_at":"2025-03-08T19:10:04Z","title":"DOFA-CLIP: Multimodal Vision-Language Foundation Models for Earth Observation","version":2},"cited_work":{"arxiv_id":"2503.06312","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.06312","snapshot_observed_at":"2026-07-10T19:07:35.356549Z","title":"Dofa-clip: Multimodal vision-language foundation models for earth observation","venue":"cs.CV","work_id":"7f42dcae-6d9d-401f-b5dc-f6cfe17a20a8","year":2025},"citing_paper":{"arxiv_id":"2607.07758","last_updated":"2026-07-08T14:31:02Z","snapshot_observed_at":"2026-08-20T20:31:31.079919Z","submitted_at":"2026-07-08T14:31:02Z","title":"Scalable and Trustworthy Earth Observation Foundation Models","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-07-10T18:58:32.854964Z"},"links":{"cited_paper":"/paper/2503.06312","citing_paper":"/paper/2607.07758"},"observation_digest":"sha256:34bbeb87c0c84c638bb202e9fe1bc74d53bd90056fec4a967b6a94d6ff125637","observation_id":"27a3a62e-56da-4ee3-8ca9-97f136dae72a","resolution":{"observed_at":"2026-07-10T19:07:35.358051Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2503.06312/citation-record","integrity":"/paper/2503.06312/integrity","json":"/paper/2503.06312/citation-record.json","paper":"/paper/2503.06312"},"outbound":[],"paper":{"arxiv_id":"2503.06312","last_updated":"2025-07-22T15:05:39Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-18T20:11:07.112891Z","submitted_at":"2025-03-08T19:10:04Z","title":"DOFA-CLIP: Multimodal Vision-Language Foundation Models for Earth Observation"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 7 inbound Pith citation observations for arXiv:2503.06312."}