{"as_of":"2026-08-10T02:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e2786166f461bf26ccace98f558788422bc6a688bd4f7b1c09d12b358e6f4fb3","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":21,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":21,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":21,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T20:36:27.109914Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T15:58:38.244809Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2401.14159","last_updated":"2024-01-25T13:12:09Z","snapshot_observed_at":"2026-07-06T17:20:25.138890Z","submitted_at":"2024-01-25T13:12:09Z","title":"Grounded SAM: Assembling Open-World Models for Diverse Visual Tasks","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-11T06:20:15.656356Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2401.14159"},"observation_digest":"sha256:8f7155cb892a2f8acc28a98b4a4392d33cd8b378163db736a39a0cc0d4399046","observation_id":"f77e605a-4257-48f1-bf6e-74856ab78195","resolution":{"observed_at":"2026-05-11T06:20:16.231345Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2402.00253","last_updated":"2024-05-06T01:10:01Z","snapshot_observed_at":"2026-07-06T17:23:26.913214Z","submitted_at":"2024-02-01T00:33:21Z","title":"A Survey on Hallucination in Large Vision-Language Models","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-13T22:10:10.186950Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2402.00253"},"observation_digest":"sha256:5b0a0b27fe46640f71cc9f238fe967cc01f0cc25c691b0a483b2d43476f2b9e0","observation_id":"2bb2d1cd-a897-4e51-8ef6-a3cfab578efc","resolution":{"observed_at":"2026-05-13T22:10:10.262114Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-09T20:36:27.109914Z","title":"Recognize anything: A strong image tagging model,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.19331","last_updated":"2025-01-31T17:31:19Z","snapshot_observed_at":"2026-08-10T02:27:20.743382Z","submitted_at":"2025-01-31T17:31:19Z","title":"Consistent Video Colorization via Palette Guidance","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-09T20:36:27.109914Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2501.19331"},"observation_digest":"sha256:b22c3968f9d0cf9fef53897ad6285b29ff04e498d48feec82749bd56e2704ce6","observation_id":"4351ae15-7939-46f5-93d9-5f9415e714d2","resolution":{"observed_at":"2026-08-09T20:36:27.109914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-09T11:52:15.003082Z","title":"Recognize anything: A strong image tagging model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.02548","last_updated":"2025-04-14T18:27:02Z","snapshot_observed_at":"2026-08-09T11:44:56.080677Z","submitted_at":"2025-02-04T18:18:50Z","title":"Mosaic3D: Foundation Dataset and Model for Open-Vocabulary 3D Segmentation","version":2},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-08-09T11:52:15.003082Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2502.02548"},"observation_digest":"sha256:a273b304dbfb99da0b05139f108e8e47c04a5f46bbe2b1e55b0fc74a6f150018","observation_id":"545cf9cc-3c2d-4110-9fd7-fa9c111dc442","resolution":{"observed_at":"2026-08-09T11:52:15.003082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2503.07598","last_updated":"2025-03-11T06:44:25Z","snapshot_observed_at":"2026-07-06T20:50:05.070886Z","submitted_at":"2025-03-10T17:57:04Z","title":"VACE: All-in-One Video Creation and Editing","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-16T00:53:53.855965Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2503.07598"},"observation_digest":"sha256:539c1db87164c74ee528c6c4606dfdb58686a2513beaf99aa2727825295e8fb2","observation_id":"30e699ae-cadf-4b19-8586-7092962bdd79","resolution":{"observed_at":"2026-05-16T00:53:54.113310Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2503.07878","last_updated":"2026-04-29T20:30:35Z","snapshot_observed_at":"2026-08-03T14:12:26.604832Z","submitted_at":"2025-03-10T21:50:58Z","title":"A Woman with a Knife or A Knife with a Woman? Measuring Directional Bias Amplification in Image Captions","version":5},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-22T23:54:15.135595Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2503.07878"},"observation_digest":"sha256:8532f37b8eddb2714e62e5594be98396291ea460ed80e02e28b84e7b4447ce0f","observation_id":"db08e31d-29fd-4975-a579-a9bb6949759d","resolution":{"observed_at":"2026-05-22T23:55:15.097067Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2504.17761","last_updated":"2025-07-31T05:05:45Z","snapshot_observed_at":"2026-07-06T21:14:24.007428Z","submitted_at":"2025-04-24T17:25:12Z","title":"Step1X-Edit: A Practical Framework for General Image Editing","version":5},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-11T14:36:41.467429Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2504.17761"},"observation_digest":"sha256:7c668a33d27d1806725bf04df3d20b837caed82a20f0f701f62b09045cf25673","observation_id":"36865902-00ca-4044-89f6-bdc67c795e50","resolution":{"observed_at":"2026-05-11T14:36:41.924000Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-07T14:14:35.758398Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19569","last_updated":"2025-05-26T06:33:48Z","snapshot_observed_at":"2026-08-07T14:09:18.072135Z","submitted_at":"2025-05-26T06:33:48Z","title":"What You Perceive Is What You Conceive: A Cognition-Inspired Framework for Open Vocabulary Image Segmentation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T14:14:35.758398Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2505.19569"},"observation_digest":"sha256:5788dca400021269b991c005d9fd1fbfd0f7ae7595a002cb1db1b4ab191e06f2","observation_id":"35cdf453-fea3-4095-8941-1fcc73ad8d74","resolution":{"observed_at":"2026-08-07T14:14:35.758398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-07T05:03:16.194374Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08968","last_updated":"2025-06-10T16:41:33Z","snapshot_observed_at":"2026-08-07T13:57:35.713590Z","submitted_at":"2025-06-10T16:41:33Z","title":"ADAM: Autonomous Discovery and Annotation Model using LLMs for Context-Aware Annotations","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T05:03:16.194374Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2506.08968"},"observation_digest":"sha256:3f3f82d3f1813f578249ad13e1e5b937ad00ac82f9158adefbdc22ecabd197b5","observation_id":"baae690b-501f-4003-bd4b-b06b3901c93b","resolution":{"observed_at":"2026-08-07T05:03:16.194374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-06T16:58:56.201780Z","title":"Recognize Anything: A Strong Image Tagging Model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12123","last_updated":"2025-07-16T10:47:12Z","snapshot_observed_at":"2026-08-06T16:50:58.569190Z","submitted_at":"2025-07-16T10:47:12Z","title":"Open-Vocabulary Indoor Object Grounding with 3D Hierarchical Scene Graph","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T16:58:56.201780Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2507.12123"},"observation_digest":"sha256:1746b8f004c3606961e82e75b8d4e1e9a034f10388248f52636d3152a5e6a825","observation_id":"5f0f4907-fe50-4cc2-89aa-f235015a8176","resolution":{"observed_at":"2026-08-06T16:58:56.201780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-05T12:27:05.091735Z","title":"Recognize anything: A strong image tagging model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.01656","last_updated":"2025-09-01T17:57:49Z","snapshot_observed_at":"2026-08-09T16:17:17.533156Z","submitted_at":"2025-09-01T17:57:49Z","title":"Reinforced Visual Perception with Tools","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-05T12:27:05.091735Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2509.01656"},"observation_digest":"sha256:ab3c8dd4ceb4533c705878a10e05fbac1418f27a99cbbaed512d0002f13779fd","observation_id":"fff8ffb9-0e83-4198-b399-5c03635fc775","resolution":{"observed_at":"2026-08-05T12:27:05.091735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2604.18562","last_updated":"2026-04-22T02:31:50Z","snapshot_observed_at":"2026-08-04T10:02:00.731719Z","submitted_at":"2026-04-20T17:49:22Z","title":"AnchorSeg: Language Grounded Query Banks for Reasoning Segmentation","version":3},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-10T05:10:44.608959Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2604.18562"},"observation_digest":"sha256:62d0afd565f99c32ff43b36e837e1ee537bdbe4883ffda34bc5f87937809beb0","observation_id":"be448d09-7cc2-4571-bb94-51e5a993bebc","resolution":{"observed_at":"2026-05-10T09:43:49.396778Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2604.19192","last_updated":"2026-04-23T06:35:35Z","snapshot_observed_at":"2026-07-06T23:05:53.378585Z","submitted_at":"2026-04-21T07:59:36Z","title":"Empowering NPC Dialogue with Environmental Context Using LLMs and Panoramic Images","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-10T01:45:42.756787Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2604.19192"},"observation_digest":"sha256:a7f83876db8fbad2fe3b174d3eb2f1487e2176f3e91af2dcf3d932baa140ec38","observation_id":"a95c5bef-7906-4e3e-9d71-a8682a003bf3","resolution":{"observed_at":"2026-05-11T13:26:03.724298Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2604.21915","last_updated":"2026-04-23T17:57:28Z","snapshot_observed_at":"2026-07-06T23:08:23.939574Z","submitted_at":"2026-04-23T17:57:28Z","title":"Vista4D: Video Reshooting with 4D Point Clouds","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-09T22:07:10.070757Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2604.21915"},"observation_digest":"sha256:37115fb57e17ead2e9883ff3bd7f6a0466eee0d72675e07caad4a6e2c50dcdc0","observation_id":"1bca43ee-80b4-4498-837c-8f35d9944663","resolution":{"observed_at":"2026-05-11T14:16:26.263787Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2605.07141","last_updated":"2026-05-08T02:20:40Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:20:40Z","title":"Qwen3-VL-Seg: Unlocking Open-World Referring Segmentation with Vision-Language Grounding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-11T02:35:57.843351Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2605.07141"},"observation_digest":"sha256:80823c4a033ba1361074c7f5f11cd0f0e35c46925fbb474630c9533561919274","observation_id":"28b87019-9fae-4f06-9162-8828411f7cb8","resolution":{"observed_at":"2026-05-11T03:10:53.994295Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2605.21244","last_updated":"2026-05-20T14:33:13Z","snapshot_observed_at":"2026-08-07T04:02:33.199414Z","submitted_at":"2026-05-20T14:33:13Z","title":"SR-Ground: Image Quality Grounding for Super-Resolved Content","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-21T05:34:17.056685Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2605.21244"},"observation_digest":"sha256:4e3a0e9d2828303efd98b73b4fae32822358fa2ceeb587ef67dc38d26b2fed8e","observation_id":"928a528b-2563-47d9-873f-f3b0b8833e91","resolution":{"observed_at":"2026-05-21T05:34:40.136791Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2606.01072","last_updated":"2026-06-05T06:32:49Z","snapshot_observed_at":"2026-08-02T21:41:54.386640Z","submitted_at":"2026-05-31T07:34:25Z","title":"Expanding Spatial and Temporal Context for Robotic Imitation Learning With Scene Graphs","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-06-28T17:24:02.843529Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2606.01072"},"observation_digest":"sha256:4411818076ed40ab9a116ea7de39ab1b26cdff77e1908931877b4831b530a68a","observation_id":"b0d85f2a-4280-4640-87f9-8a85c30d95fc","resolution":{"observed_at":"2026-07-01T21:16:13.673065Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2606.12849","last_updated":"2026-06-11T03:24:05Z","snapshot_observed_at":"2026-08-03T10:50:52.195357Z","submitted_at":"2026-06-11T03:24:05Z","title":"SemanticXR: Low Power and Real-time Queryable Semantic Mapping with an Object-Level Device-Cloud Architecture","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-27T06:13:49.396855Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2606.12849"},"observation_digest":"sha256:1dd7651194ed8581bd4cadfbeb3775f10f945c13d7843118e19836a6ea148a32","observation_id":"37f6e3cc-df5a-4c54-afb8-df89e4b5768b","resolution":{"observed_at":"2026-07-03T15:58:38.248195Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":"2306.03514","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-03T15:58:38.244809Z","title":"Recognize anything: A strong image tagging model.arXiv preprint arXiv:2306.03514","venue":null,"work_id":"cd2a8430-e48f-4b86-99fb-64d9a6f5556b","year":2023},"citing_paper":{"arxiv_id":"2606.28592","last_updated":"2026-06-26T20:36:26Z","snapshot_observed_at":"2026-08-08T04:44:34.033813Z","submitted_at":"2026-06-26T20:36:26Z","title":"Embodiment Meets Environment: Toward Context-Aware, Safe Physical Caregiving Robots","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-30T00:56:51.666226Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2606.28592"},"observation_digest":"sha256:dea12ff4c2d672d64fce78d89cffb99eb1fad94a44b7fdabe89f4c2cb31fa0ae","observation_id":"c6f0ebdf-5b56-4461-809a-c50d35f44a99","resolution":{"observed_at":"2026-07-01T15:55:49.243148Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-07-11T16:26:39.443485Z","title":"arXiv preprint arXiv:2306.03514 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04612","last_updated":"2026-07-06T02:37:22Z","snapshot_observed_at":"2026-08-06T16:33:39.351941Z","submitted_at":"2026-07-06T02:37:22Z","title":"StructuredEdit: Constraint-Aware Graphic Design Editing via Differentiable Parameter Propagation","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-07-11T16:26:39.443485Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2607.04612"},"observation_digest":"sha256:97b5f5d2b105ffc20c601012e00e4e1c03c60fe2ad7bb9e6c30c5cb85cd93b2c","observation_id":"dc268afb-10ac-4d4c-8bcc-ab301db38d8a","resolution":{"observed_at":"2026-07-11T16:26:39.443485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03514","snapshot_observed_at":"2026-08-01T22:34:46.291494Z","title":"arXiv preprint arXiv:2306.03514 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15711","last_updated":"2026-07-17T07:49:05Z","snapshot_observed_at":"2026-08-09T11:27:36.259203Z","submitted_at":"2026-07-17T07:49:05Z","title":"Efficient Difficulty-Aware Dynamic Routing for Diffusion-Based Real-World Image Super-Resolution","version":1},"reference_index":141,"source":"arxiv_source","source_observed_at":"2026-08-01T22:34:46.291494Z"},"links":{"cited_paper":"/paper/2306.03514","citing_paper":"/paper/2607.15711"},"observation_digest":"sha256:8b6a88dd2927e1224a493ce11f9c9832e494665f27c1ff6122ff7110ce1a8a61","observation_id":"bc185dca-cc97-409c-be44-371d52aaa6eb","resolution":{"observed_at":"2026-08-01T22:34:46.291494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2306.03514/citation-record","integrity":"/paper/2306.03514/integrity","json":"/paper/2306.03514/citation-record.json","paper":"/paper/2306.03514"},"outbound":[],"paper":{"arxiv_id":"2306.03514","last_updated":"2023-06-09T15:21:06Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T15:39:06.851333Z","submitted_at":"2023-06-06T09:00:10Z","title":"Recognize Anything: A Strong Image Tagging Model"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 21 inbound Pith citation observations for arXiv:2306.03514."}