{"as_of":"2026-08-08T08:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e33ba0fe2d06558ca53a160d946eb97ba8b2bbb1c1d1665a5aec9f36c43bc625","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:47:27.120881Z","state":"measured"},{"denominator":60,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":60,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.17695/citation-record","integrity":"/paper/2505.17695/integrity","json":"/paper/2505.17695/citation-record.json","paper":"/paper/2505.17695"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T14:47:20.053296Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.053296Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:18449f621ca039ba5c5bb673bcda17a24a3cc08e50f21f44c57fb12e8ceeeb19","observation_id":"18aa0765-f4fc-43f3-9784-94f3d02bdcaf","resolution":{"observed_at":"2026-08-07T14:47:20.053296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:40.042876Z","title":"Vqa: Visual question answering","venue":null,"work_id":"7dea7b19-251b-40e6-9144-b5196c11b1a5","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.224226Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:f79c2ad67f6d7fe6faba5c2908f57da39a21bf69b0df7954ddd2c3f400ef2c20","observation_id":"db9ff230-e126-4178-a790-947e329204ee","resolution":{"observed_at":"2026-08-07T14:47:40.174662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:39.906864Z","title":"Coco- stuff: Thing and stuff classes in context","venue":null,"work_id":"5a608b8d-202e-4ec7-9d0f-9848b2aa0d6d","year":2018},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.354744Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:ee8cd4a9302330dc5a466cce52ac43a393cf0f7bdaee598a3d37e9273d384c15","observation_id":"75d2d425-9aed-44f3-bff1-77c513afcb5e","resolution":{"observed_at":"2026-08-07T14:47:39.965237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:39.771808Z","title":"Detect what you can: De- tecting and representing objects using holistic models and body parts","venue":null,"work_id":"9f6e34ec-1ccf-4aae-bf31-8d26462a8d9a","year":1971},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.582089Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:3919e403ddb44ab30578e00708de7baa01fcd6012de44121027359c5dbdb8f62","observation_id":"0040dbe8-b07a-4196-b433-0c795bcf9153","resolution":{"observed_at":"2026-08-07T14:47:39.855083Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:39.504157Z","title":"Sam4mllm: Enhance multi- modal large language model for referring expression seg- mentation","venue":null,"work_id":"cfb634e8-de9b-493f-a7bf-a1394f5473a1","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.719249Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:61522756d3d5086223b4923dabe04dccbcc6a31b872c31064e8adb68695fcd2e","observation_id":"29e63728-7e7c-4d1d-840a-d251c80e6c04","resolution":{"observed_at":"2026-08-07T14:47:39.662221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:39.284318Z","title":"The cityscapes dataset for semantic urban scene understanding","venue":null,"work_id":"0b0c2ae4-9e1a-4dfe-a2f9-6a2c2683a8d9","year":2016},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.814327Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:46be9d89e59ab97b3c6f19fd5906e2be6e93f36927435fbe3d18a2348ebc8de1","observation_id":"593d3b04-b604-4091-b064-ecf33998db1b","resolution":{"observed_at":"2026-08-07T14:47:39.388987Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:38.992390Z","title":"Divergen: Improving instance segmentation by learning wider data distribution with more diverse generative data","venue":null,"work_id":"44bf5f59-6b9d-44bf-9f90-bf9abcba0c3f","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:20.945150Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:a14488289c0967a1f53101bcfc2d332ce3dcff6a460f4c96011f366ee1cfc358","observation_id":"ad7a91b9-c475-43c5-a60d-ee70e36b05ab","resolution":{"observed_at":"2026-08-07T14:47:39.171952Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:38.764887Z","title":"Finding nemo: Negative- mined mosaic augmentation for referring image segmentation","venue":null,"work_id":"855726e2-aef2-47c2-8127-f8e6f49e8822","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.055373Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:fa888434e62d1550fbf61fd87a76a8a18b5bdde6e9765ceb6e706afe21ffd07c","observation_id":"4f51af55-ece7-4008-92f8-d16910be9951","resolution":{"observed_at":"2026-08-07T14:47:38.874547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:38.459396Z","title":"Mixgen: A new multi- modal data augmentation","venue":null,"work_id":"f7db3447-aab3-4b28-afc5-71b6822adf1a","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.156345Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:8cb211360c1084a99f52cb0078e2b10d8a3681e777086c93ebc32058b18b283c","observation_id":"ed78cdbe-d00e-4fc0-ba65-b58e6c349938","resolution":{"observed_at":"2026-08-07T14:47:38.621618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:38.220948Z","title":"Partimagenet: A large, high-quality dataset of parts","venue":null,"work_id":"4f3e6a6c-7696-458f-8571-5b87f570a90d","year":2022},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.270993Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:14096b99a0300742d7aee4c71bc0170d938955ceb45d94cbb296f9394b3a5163","observation_id":"6b5d6fab-f880-40aa-ab36-6cfdcd249263","resolution":{"observed_at":"2026-08-07T14:47:38.344668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:21.417306Z","title":"Denoising dif- fusion probabilistic models","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.417306Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:50a5cd5befb60d1beb024f8b58c2ea939e633c5993803f15475d64bbcbbf11df","observation_id":"4d663812-4d0e-492c-a258-9c9a1b0d5a54","resolution":{"observed_at":"2026-08-07T14:47:21.417306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:37.955024Z","title":"Beyond one-to-one: Rethinking the referring image segmentation","venue":null,"work_id":"ea54812d-f07f-42ff-9d89-473d717230b4","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.520435Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:5fab751da7ec6f0cfc202ce9be1a0efb1eeea72727f54f4745bc3a487ddf4fd5","observation_id":"cc5a2dd0-2c42-4778-800d-c7a349015700","resolution":{"observed_at":"2026-08-07T14:47:38.073131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T14:47:21.632881Z","title":"Gpt-4o system card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.632881Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:f986a9f6cdd97095d7b65f46a1f1904981842a76e1c9815cf266505c164d1b6a","observation_id":"982feeb2-8e80-48da-ad7e-2911838a7c6b","resolution":{"observed_at":"2026-08-07T14:47:21.632881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.10086","last_updated":"2024-08-19T15:27:25Z","snapshot_observed_at":"2026-07-06T19:02:56.021845Z","submitted_at":"2024-08-19T15:27:25Z","title":"ARMADA: Attribute-Based Multimodal Data Augmentation","version":1},"cited_work":{"arxiv_id":"2408.10086","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.10086","snapshot_observed_at":"2026-08-07T14:47:27.699275Z","title":"ARMADA: Attribute-Based Multimodal Data Augmentation","venue":"cs.AI","work_id":"3e5693df-934d-4760-b4d7-9688523d2832","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.743339Z"},"links":{"cited_paper":"/paper/2408.10086","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:12c5b381847ba3b45d24411bd54309692e059f314abf0010459440e120c8bd67","observation_id":"04aa2b42-bb28-45f9-991e-09beeba581d7","resolution":{"observed_at":"2026-08-07T14:47:27.779282Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:37.730601Z","title":"Referitgame: Referring to objects in pho- tographs of natural scenes","venue":null,"work_id":"307c5fa1-43eb-40f7-9b39-07ffc2118017","year":2014},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.838974Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:16ef735a2d7bf7ab04b1d1a93eddb06e75852804e49def17b3b097a963c46e83","observation_id":"81d70f02-e4d9-499e-a0d0-6180e3306afe","resolution":{"observed_at":"2026-08-07T14:47:37.826592Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:37.497452Z","title":"Segment any- thing","venue":null,"work_id":"67fcfcb0-ac6b-4de1-b020-9ce7be45f241","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:21.921379Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:8ad4712c55389ed701cc11c331753b4393228f06049bfefe8f6dd8b638095e14","observation_id":"ae266e83-b34f-4ec4-97a8-e409dbdb1892","resolution":{"observed_at":"2026-08-07T14:47:37.601437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:37.208877Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":"ed9e5410-f4bd-4a1f-8415-08d25013ec93","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.006435Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:a4498dd85d49630664050c8febf5beedc3160a57d21105b53c698c3ea7c21bc5","observation_id":"1bacd787-1def-49b8-b1b3-46cb72466998","resolution":{"observed_at":"2026-08-07T14:47:37.363436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:36.953305Z","title":"Bigdatasetgan: Synthe- sizing imagenet with pixel-wise annotations","venue":null,"work_id":"47cf7577-8635-439c-9dda-0355d515b342","year":2022},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.120552Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:81de38f6e0008d9d8685d87f2b6aeb46c5e1259d43e8426c379b485f7a3cded6","observation_id":"698c7ee5-b400-4e3a-9262-594cd968a8ac","resolution":{"observed_at":"2026-08-07T14:47:37.057411Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:36.733177Z","title":"Gres: Gener- alized referring expression segmentation","venue":null,"work_id":"798c458f-f8b4-49bb-af02-0d99c3c975e9","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.239095Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:a73c3ad5dfbb438ad969ece7e48613782683b5003540251a6a107634ff0285f5","observation_id":"b7c1f8cc-3b97-41dd-b890-b773d04e95ec","resolution":{"observed_at":"2026-08-07T14:47:36.829544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:36.489300Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":"1fe37c8c-c332-4eb1-a0fb-b44a2ecda485","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.322938Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:5263c9cc7d0133f96e80c6b6fd42987cf99c389a291978e84b8b3b2907220656","observation_id":"f3e6ca16-c619-4cac-8b4d-c154571f153c","resolution":{"observed_at":"2026-08-07T14:47:36.591923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:36.250363Z","title":"Visual instruction tuning","venue":null,"work_id":"381f82a4-9954-4de7-9f24-9b1c66d7b403","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.405143Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:0a58268bb43a264dbd86c46475dd0c26110bebdc2d8393dc2ca83e146bb72dc3","observation_id":"e1829636-215f-4de0-8a45-31ebf64feeab","resolution":{"observed_at":"2026-08-07T14:47:36.367977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:35.981959Z","title":"Learning multimodal data augmentation in feature space","venue":null,"work_id":"ca34b59f-af25-45ae-871b-d748ce19d70f","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.497171Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:4a0d2f94733d7f131e0e86f5e617364c52917554a1e3e8fe2a365390fa400821","observation_id":"7b0b0c59-ebbc-4ddc-bb91-8141aea95786","resolution":{"observed_at":"2026-08-07T14:47:36.085520Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:35.621367Z","title":"Generation and com- prehension of unambiguous object descriptions","venue":null,"work_id":"6999331f-867b-4958-9488-aa38c69cbe5c","year":2016},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.582842Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:1016763074406537c25a0513089cd907c392bc37375e6b45bc13d8527c577bc6","observation_id":"2b756d41-74bb-40b7-8fa0-0565df969ac0","resolution":{"observed_at":"2026-08-07T14:47:35.769701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:35.375128Z","title":"Arm- bench: An object-centric benchmark dataset for robotic ma- nipulation","venue":null,"work_id":"29883de5-3c2f-4945-97d4-f171491c486b","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.655738Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:e65eb0be4860bf8b30192501dafb01a17f981642ffb0ba3a2584aed44c33c3f5","observation_id":"2d0486e6-a933-42fd-a5ad-faba589ad9f0","resolution":{"observed_at":"2026-08-07T14:47:35.465501Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:35.111647Z","title":"The mapillary vistas dataset for semantic understanding of street scenes","venue":null,"work_id":"983f3a66-7956-46d2-b5e3-242c956863b9","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.831522Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:042faa2b9ec91a0998d4059c77f14d992814a1377c47df468b3f38861256ac81","observation_id":"376cf2ff-cf1e-4986-b3dc-81d7c8a7364a","resolution":{"observed_at":"2026-08-07T14:47:35.253944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:34.812467Z","title":"Dataset diffusion: Diffusion-based synthetic data generation for pixel-level semantic segmentation","venue":null,"work_id":"bf0ec2a9-e27d-497f-8d01-d29cf7885ecd","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:22.906508Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:9636bb8aae6a8ce991650a64789c25336b01a7e897086c971593f5904eaebd45","observation_id":"8c7a92a1-16bf-4206-a287-7634dca742da","resolution":{"observed_at":"2026-08-07T14:47:34.982424Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:34.569940Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"75a9c6f8-e9c7-4cc6-acc7-8917e430a2ef","year":2021},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.020067Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:18b5c3062fe69923851da89e3b21d2908addf98a78bcf21ba74f3492441d1aac","observation_id":"6e601524-4404-4457-948b-7ebddc73b350","resolution":{"observed_at":"2026-08-07T14:47:34.697135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:34.235391Z","title":"Paco: Parts and attributes of common objects","venue":null,"work_id":"47e10bbf-9008-4f0d-8258-15b33a577e0e","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.143071Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:f937a8f02bf6c8ff53609dd41874e668ee612a475bb338727811c4288b988bc7","observation_id":"e2e5d3db-c511-4199-8a54-e25671517fc1","resolution":{"observed_at":"2026-08-07T14:47:34.330678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:33.921262Z","title":"Glamm: Pixel grounding large multimodal model","venue":null,"work_id":"8fbe57ad-4363-45ec-91a5-1c280531d20c","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.247935Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:8f81fde485727d30f7d2457dea8b8bfa14554d2e3582522acc1c166c8ea76566","observation_id":"fa6d6adf-7a56-43be-8858-94a792959ffd","resolution":{"observed_at":"2026-08-07T14:47:34.086363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:33.602571Z","title":"SAM 2: Segment anything in images and videos","venue":null,"work_id":"9a2869d3-9d63-4c53-a46a-40de4e84f0bf","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.370944Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:b11f3b746fc35a55a21e31056bc5fbd9cdad978ae94c220e5f594c9d993d94af","observation_id":"86e7719c-f194-4a07-ab95-adc4a619e2ed","resolution":{"observed_at":"2026-08-07T14:47:33.780743Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:33.381905Z","title":"Pixellm: Pixel reasoning with large multimodal model","venue":null,"work_id":"ba7d32b0-91e5-415f-b93c-43fae616bd17","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.476000Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:6ad448f11fa2ad0b282d158cabcd2e320de898ba7215bfccbc311c9c3745c91b","observation_id":"c4b4b0f9-a4a7-4e5d-bbd5-950cb452100a","resolution":{"observed_at":"2026-08-07T14:47:33.485800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:33.135262Z","title":"Grounding of textual phrases in images by reconstruction","venue":null,"work_id":"42cd932a-f0fa-493c-9a0d-e28b3b86fb9e","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.585827Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:6d333220e16031837ffe67a0f9ab4da434128eee887ab9c7031863042c45bcbc","observation_id":"9cb64232-9601-4aaa-8e40-c94bc1fd775a","resolution":{"observed_at":"2026-08-07T14:47:33.238447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1805.00123","last_updated":"2018-04-30T22:49:54Z","snapshot_observed_at":"2026-07-06T06:36:33.930118Z","submitted_at":"2018-04-30T22:49:54Z","title":"CrowdHuman: A Benchmark for Detecting Human in a Crowd","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1805.00123","snapshot_observed_at":"2026-08-07T14:47:23.699161Z","title":"Crowdhuman: A bench- mark for detecting human in a crowd","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.699161Z"},"links":{"cited_paper":"/paper/1805.00123","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:47a131d5be0f5497b88417864c04a94d2b864b120fbf1de38eb8927400f886cb","observation_id":"a7751b07-0325-4461-8730-f2f0486faf37","resolution":{"observed_at":"2026-08-07T14:47:23.699161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:32.909721Z","title":"Denoising diffusion implicit models","venue":null,"work_id":"afda224a-8d74-4953-af9b-531e7afb4742","year":2021},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.808606Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:837113542680db8a645a9d2d290f9da449fb82c32247bb0ed36dcecba6dbbc82","observation_id":"a09e70b1-3ba4-4121-a076-ae22b72a40aa","resolution":{"observed_at":"2026-08-07T14:47:33.015018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.02048","last_updated":"2025-05-28T08:11:16Z","snapshot_observed_at":"2026-07-06T20:16:22.788626Z","submitted_at":"2025-01-03T19:00:00Z","title":"DreamMask: Boosting Open-vocabulary Panoptic Segmentation with Synthetic Data","version":2},"cited_work":{"arxiv_id":"2501.02048","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.02048","snapshot_observed_at":"2026-08-07T14:47:27.478498Z","title":"DreamMask: Boosting Open-vocabulary Panoptic Segmentation with Synthetic Data","venue":"cs.CV","work_id":"98fc4bb5-4763-48b9-8b16-6c2dee21daf2","year":2025},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:23.916439Z"},"links":{"cited_paper":"/paper/2501.02048","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:5b5571a2b93bf03b2721c510fbbcd519acd977f8f6a8ca156acca51e3dee7f3c","observation_id":"5f946cb7-fbce-46cf-8317-c2e35efda692","resolution":{"observed_at":"2026-08-07T14:47:27.583189Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:32.636223Z","title":"Cris: Clip-driven referring image segmentation","venue":null,"work_id":"7cb59edd-5241-419b-874b-8301b93a9a85","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.069836Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:0377e39c3210a3b76c9968a0468c2443fe17790d2a78430dd0e0c713d32ecac9","observation_id":"110b8971-dade-48ef-83e0-968aa9e3a808","resolution":{"observed_at":"2026-08-07T14:47:32.788625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01330","last_updated":"2023-10-02T16:48:50Z","snapshot_observed_at":"2026-08-03T20:58:17.113036Z","submitted_at":"2023-10-02T16:48:50Z","title":"Towards reporting bias in visual-language datasets: bimodal augmentation by decoupling object-attribute association","version":1},"cited_work":{"arxiv_id":"2310.01330","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.01330","snapshot_observed_at":"2026-08-07T14:47:27.261656Z","title":"Towards reporting bias in visual-language datasets: bimodal augmentation by decoupling object-attribute association","venue":"cs.CV","work_id":"e10fd36f-9ce4-4e6b-bb0f-a4130b72b569","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.223505Z"},"links":{"cited_paper":"/paper/2310.01330","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:5b6cc7815ee11ad08af50a18a528ed9f36a594d9f346cc6286a0744cf55d949c","observation_id":"af2dc787-bdba-49e3-aa9e-79ca09ac039e","resolution":{"observed_at":"2026-08-07T14:47:27.366597Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:32.399025Z","title":"Diffumask: Synthesizing images with pixel-level annotations for semantic segmentation using diffu- sion models","venue":null,"work_id":"56165b03-9266-4782-8ef6-df04ef90a59e","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.377245Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:bae4a3f8b46627283ae5d4ca78d8995be88894e81f5cffc7d7e66fec259fa8cf","observation_id":"311c22a3-2dc1-427d-811d-a8c2a2111c5b","resolution":{"observed_at":"2026-08-07T14:47:32.549960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:32.169723Z","title":"Gsva: Generalized segmentation via multimodal large language models","venue":null,"work_id":"189931f4-0e13-42e1-9032-81ca4ef8a268","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.527351Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:90f872255d4a59cd39a675420d32ae183c134082566a0deafe542433b71f8bc2","observation_id":"65cc8935-735d-46fc-a756-96c111073256","resolution":{"observed_at":"2026-08-07T14:47:32.290075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10629","last_updated":"2024-10-20T14:35:31Z","snapshot_observed_at":"2026-07-06T19:33:08.264958Z","submitted_at":"2024-10-14T15:36:42Z","title":"SANA: Efficient High-Resolution Image Synthesis with Linear Diffusion Transformers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10629","snapshot_observed_at":"2026-08-07T14:47:24.674218Z","title":"Sana: Efficient high-resolution image synthesis with lin- ear diffusion transformers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.674218Z"},"links":{"cited_paper":"/paper/2410.10629","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:62e170c6df617cfe0853b49e2fb42e621d93f67419eefaa26da16f86a078cec6","observation_id":"664af2dd-7586-4e2f-b56f-a4755bf3bebb","resolution":{"observed_at":"2026-08-07T14:47:24.674218Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:31.960808Z","title":"Mosaicfusion: Diffusion models as data augmenters for large vocabulary instance segmentation","venue":null,"work_id":"08bee63e-2ba2-4cf0-bb72-fba12dc841b9","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.804534Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:97f378d9299c07a947a01515aff6d655c89fbda22d916556e8955428549459fc","observation_id":"f9c2fbe1-7532-40b5-a8b8-a64bd7d7408e","resolution":{"observed_at":"2026-08-07T14:47:32.060147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:31.732033Z","title":"Bridging vision and language encoders: Parameter-efficient tuning for referring image segmentation","venue":null,"work_id":"3a8b7254-c13e-42ac-8e0a-1189616da876","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:24.935045Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:4ddc3307b04f3741a4c529379dc8eddf7b751a1f4d81ffce3cb403a957cea1b9","observation_id":"896b32b1-bbf4-4502-85ab-79dc11b0e4f7","resolution":{"observed_at":"2026-08-07T14:47:31.877017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:31.466874Z","title":"Panoptic scene graph gen- eration","venue":null,"work_id":"6b88279d-ee3c-41a7-9c30-72941647e486","year":2022},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.078638Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:ac91a11d376f1d031d8d765d345097db5db8b31efcdef12c8552b089b526b999","observation_id":"d107d3a5-db7d-4cc3-b2c0-5f562362d384","resolution":{"observed_at":"2026-08-07T14:47:31.620806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:31.182229Z","title":"Freemask: Synthetic images with dense annotations make stronger segmentation models","venue":null,"work_id":"239f77d1-7a2c-4639-81b7-5cf90a1429a4","year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.211282Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:dea7d3b5fb08a72124b5c7c3754077224b86b8ca470e53eb78d7076ee15da3df","observation_id":"1365327d-86c0-4d4a-b647-c82600da3a52","resolution":{"observed_at":"2026-08-07T14:47:31.349947Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:30.896671Z","title":"Lavt: Language-aware vi- sion transformer for referring image segmentation","venue":null,"work_id":"75e0ea6e-8de6-427e-8130-af11c8cdb849","year":2022},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.303573Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:c856aea3ac5cd1c25c166319dc40d037891a795d77a16f15d1e42548837da342","observation_id":"b8f52a39-dab2-41d0-8719-3e2d99ac9dbc","resolution":{"observed_at":"2026-08-07T14:47:31.049158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:30.642365Z","title":"Seggen: Supercharging segmentation models with text2mask and mask2img synthesis","venue":null,"work_id":"caba22cf-16d4-4132-85bb-4d4306b0d527","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.460160Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:9648626bd710e59c92b53de5c5f4325dbc8d1e039520a3b4d7f7357eb175d6c5","observation_id":"5b90db18-c264-45a0-a000-dc216eddebb7","resolution":{"observed_at":"2026-08-07T14:47:30.759804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13893","last_updated":"2025-01-23T18:08:57Z","snapshot_observed_at":"2026-07-06T20:25:08.527025Z","submitted_at":"2025-01-23T18:08:57Z","title":"Pix2Cap-COCO: Advancing Visual Comprehension via Pixel-Level Captioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13893","snapshot_observed_at":"2026-08-07T14:47:25.624443Z","title":"Pix2cap-coco: Advancing visual comprehension via pixel-level captioning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.624443Z"},"links":{"cited_paper":"/paper/2501.13893","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:cecbdbe5f3daebe8a21190fbf33231eeeb808d640f3865c7e143bdcee448cb64","observation_id":"9e21ba7b-94d9-4cd1-8b4c-13e978fbf85c","resolution":{"observed_at":"2026-08-07T14:47:25.624443Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:30.395767Z","title":"From image descriptions to visual denotations: New similarity metrics for semantic inference over event descrip- tions","venue":null,"work_id":"6177647f-ed69-404d-9d7b-65f270573dc1","year":2014},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.738988Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:fbb74f189b70484ac94694a46af2106bcf7a5028fa7856edfa34faf5b347be27","observation_id":"be04d225-2071-4d7c-9095-03a4d872127f","resolution":{"observed_at":"2026-08-07T14:47:30.484684Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:30.140045Z","title":"Coca: Contrastive captioners are image-text foundation models","venue":null,"work_id":"b69b49da-a509-42cf-a38a-5a604d740f54","year":2022},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.851673Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:b3872f24e1d9c011d28e21e2d050071be34f8179e49c083ae6b3e3d985ba512f","observation_id":"acee422e-2b11-4869-b89b-d5faaecd52b9","resolution":{"observed_at":"2026-08-07T14:47:30.246850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:29.823972Z","title":"Modeling context in referring expressions","venue":null,"work_id":"859f06ad-be0e-4b03-bc9a-b153556b55ff","year":2016},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:25.960341Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:2d09c64c82bdcad1dfd9a9df893b482772fa0045a443b4c2f62207f786124c7e","observation_id":"298725ce-b1b7-465e-9404-4ce1c629a6b5","resolution":{"observed_at":"2026-08-07T14:47:30.004079Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:29.523382Z","title":"Pseudo- ris: Distinctive pseudo-supervision generation for referring image segmentation","venue":null,"work_id":"0b19d853-3fad-49e1-a96d-78ac80921515","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.080046Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:572cb37497ee39e27a01f615c05c0e48e7f4a7fd20b47f353f2a57e437a49762","observation_id":"53ee01ac-5595-4c72-ad2d-3706a951b1f9","resolution":{"observed_at":"2026-08-07T14:47:29.677320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:29.272982Z","title":"Revisiting counterfactual prob- lems in referring expression comprehension","venue":null,"work_id":"d682486e-39a8-4fcd-a97e-5501178e06bd","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.169375Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:98b916a8e5a773442ff0d249caddd4d67583c509909a7ac2a83a8c208fc951c0","observation_id":"029d00c9-a3b2-46ca-a571-56dae9d18660","resolution":{"observed_at":"2026-08-07T14:47:29.418166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:29.062602Z","title":"Datasetgan: Efficient labeled data factory with minimal human effort","venue":null,"work_id":"df2ae53e-1c1b-47e6-956e-b080396f8224","year":2021},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.285795Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:db6e59c341c9d6c3a765f41970cefaa3340207f69bdbe0d0a03e64ba03f2c38f","observation_id":"f413083f-ea99-4897-bd65-4ed33e1fddfe","resolution":{"observed_at":"2026-08-07T14:47:29.165630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.20076","last_updated":"2025-03-10T12:34:24Z","snapshot_observed_at":"2026-08-06T23:39:09.398813Z","submitted_at":"2024-06-28T17:38:18Z","title":"EVF-SAM: Early Vision-Language Fusion for Text-Prompted Segment Anything Model","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.20076","snapshot_observed_at":"2026-08-07T14:47:26.416689Z","title":"Evf-sam: Early vision-language fusion for text-prompted seg- ment anything model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.416689Z"},"links":{"cited_paper":"/paper/2406.20076","citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:f9b1812444355247d39c962180b9b288e22a41beb6b3eef23d989a7ecec642eb","observation_id":"eea956fa-8b55-47b8-904c-11aabd935bef","resolution":{"observed_at":"2026-08-07T14:47:26.416689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:28.801476Z","title":"Psalm: Pixelwise segmentation with large multi-modal model","venue":null,"work_id":"d4c6cb6b-3f5b-4d80-b4e4-60c2d6516b50","year":2024},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.551994Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:417a8b273488fcc9809f7a06e6f02ee45544b48bafac1607fc7cb8d97781671b","observation_id":"ca4ff209-3265-47d3-8c01-6989a655964e","resolution":{"observed_at":"2026-08-07T14:47:28.928964Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:28.588830Z","title":"X-paste: Revisiting scalable copy-paste for in- stance segmentation using clip and stablediffusion","venue":null,"work_id":"524c8727-0cb6-46e2-8fed-5daa773a5833","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.658210Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:9be84fd1400dcca7b3c9af2a233c8af265604e527c812d993b4f83fb784ac845","observation_id":"f7082026-268a-4aff-b4bf-f8b162d5061d","resolution":{"observed_at":"2026-08-07T14:47:28.685348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:26.760787Z","title":"Unleashing text-to-image diffusion models for visual perception","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.760787Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:05ef064b0ffdcc6a8b9df4c603b6dfb54e812740f031df00bce16e7ca8b2cc2a","observation_id":"faaab8c1-8e33-4b9d-bde0-3cda016368a8","resolution":{"observed_at":"2026-08-07T14:47:26.760787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:28.368649Z","title":"Scene parsing through ade20k dataset","venue":null,"work_id":"3a25d8c5-c91f-4fb6-9429-257f2cc012ef","year":2017},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:26.890631Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:76fca23cacd7d690d988c3ba6e32698a0c232620e9ac0d40ffb2e8ec6a3fe7f0","observation_id":"7a5485ab-0c7b-4bbc-a4f3-e759d859ec47","resolution":{"observed_at":"2026-08-07T14:47:28.461809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:28.071759Z","title":"Generalized decoding for pixel, image, and language","venue":null,"work_id":"c28438d4-a8f8-45ab-8e82-ccbee2efa3c3","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:27.016843Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:9e5ee33dfad03f688d471f4d5f50c92db718e66336f0fcc182dd6cd79d6f8717","observation_id":"a9a2d513-693c-4783-b193-ac6be2c89da3","resolution":{"observed_at":"2026-08-07T14:47:28.237992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:47:27.880708Z","title":"the cat sitting on the bench next to big green wooden boat in the center of the image","venue":null,"work_id":"dc5938a9-f528-4ead-a305-4ff41e0dcd53","year":2023},"citing_paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:27.120881Z"},"links":{"citing_paper":"/paper/2505.17695"},"observation_digest":"sha256:1e360617c0f7e684a85e49e9f4e472425890da3fc4287950e3c3d32ed1182484","observation_id":"926b136e-3d50-499b-99db-ad42737ca299","resolution":{"observed_at":"2026-08-07T14:47:27.953849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.17695","last_updated":"2025-05-23T10:05:16Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T14:40:03.167039Z","submitted_at":"2025-05-23T10:05:16Z","title":"SynRES: Towards Referring Expression Segmentation in the Wild via Synthetic Data"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":3,"verified_fuzzy":49},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 0 inbound Pith citation observations for arXiv:2505.17695."}