{"as_of":"2026-08-08T14:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6e3e515aea77d73b36484d285842fea01a2a4144dbb76bbc859eb61eac10f708","coverage":[{"denominator":57,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":57,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:59:44.658585Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.09612/citation-record","integrity":"/paper/2507.09612/integrity","json":"/paper/2507.09612/citation-record.json","paper":"/paper/2507.09612"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:52.030116Z","title":"Ef- ficient interactive annotation of segmentation datasets with polygon-rnn++","venue":null,"work_id":"bc783c60-78a7-4cfb-abc4-f71f8ec1801a","year":2018},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.050124Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:0ecfe3339960556ef2c9090d9e8ffa641ff0634ba94af1bea7db124c360460c8","observation_id":"68897891-50d7-4487-9c27-94c9d5301552","resolution":{"observed_at":"2026-08-06T17:59:52.095454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.904729Z","title":"Moinul Hos- sain, Gianluca Marcelli, Marc Alemany-Fornes, and Anas- tasios D","venue":null,"work_id":"17966069-5d9c-4b9b-b846-cd44c86b753f","year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.169344Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:cdebcc6adcecfbd0a1dcde5bbc6aa663638eebf3b4049f03d899e9733d93c822","observation_id":"2f1202f6-7312-49d3-af09-fceca1a60c77","resolution":{"observed_at":"2026-08-06T17:59:51.968710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.772405Z","title":"Mvtec ad–a comprehensive real-world dataset for unsupervised anomaly detection","venue":null,"work_id":"35ea596d-d8c7-4d1b-a887-af2b0c353186","year":2019},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.222264Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:8de51d9d8f60b2e5f34914566881d86a357d0fa4261d95f4fbaec64e62fd1a21","observation_id":"c6f49b15-f40c-4890-965d-70039fad130e","resolution":{"observed_at":"2026-08-06T17:59:51.811180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:40.301798Z","title":"nuscenes: A multi- modal dataset for autonomous driving","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.301798Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:866f7abb90e15e84dc1ce1311cf0d5acc8f2eb8d1852d4ab48d8c6cfdc20eb4e","observation_id":"f3e79b69-9c14-49c2-8301-6d8071ed9aef","resolution":{"observed_at":"2026-08-06T17:59:40.301798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.612281Z","title":"Focalclick: towards practical in- teractive image segmentation","venue":null,"work_id":"70b5c949-4c22-4fc3-8315-79cea1ef6444","year":2022},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.359721Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:597df8ce204e01b66652d6c0e712c741b837aaa6687e9af1b17a8f4de7ebe8e5","observation_id":"422fcf44-c7d9-4422-abe7-aea092f3e420","resolution":{"observed_at":"2026-08-06T17:59:51.664329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.474926Z","title":"FlashAttention-2: Faster attention with better par- allelism and work partitioning","venue":null,"work_id":"6f35eb9e-339a-4495-8da0-ecf87cbfa548","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.419430Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:a0542fc591491c62d8b0e6b4e9d7163fc9a4146f04c6b71ceb33a0de5df75bb7","observation_id":"c6486f07-cae1-4c13-832c-4ca56f0c00c7","resolution":{"observed_at":"2026-08-06T17:59:51.540547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.285302Z","title":"Fu, Stefano Ermon, Atri Rudra, and Christopher R´e","venue":null,"work_id":"8741f232-ca7a-4bc5-9760-11ad6f800818","year":2022},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.535287Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:0b570a9e3e621d3fba52b6b832a556b98cb83ca8680feab10e77072a69212bbd","observation_id":"1b607509-86ef-4e08-9216-2426db8d0f45","resolution":{"observed_at":"2026-08-06T17:59:51.368068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:40.649265Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.649265Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:594a4d7615d64e735e45c92dbb0a3c3da5ae3eb61ad4577c2d5d4ad81e5e3644","observation_id":"4dd43549-5bfc-4f2c-b6f5-c0203e2373ed","resolution":{"observed_at":"2026-08-06T17:59:40.649265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.07372","last_updated":"2023-11-06T13:16:00Z","snapshot_observed_at":"2026-07-06T14:30:40.323044Z","submitted_at":"2022-12-14T17:50:39Z","title":"Image Compression with Product Quantized Masked Image Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.07372","snapshot_observed_at":"2026-08-06T17:59:40.830763Z","title":"Image com- pression with product quantized masked image modeling","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.830763Z"},"links":{"cited_paper":"/paper/2212.07372","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:8868ce8d514be2abc53af78954445514d201aba25e80a6f05a49ddb91d497d00","observation_id":"949d9b8e-f720-4f8a-bd42-6d604bd9ce91","resolution":{"observed_at":"2026-08-06T17:59:40.830763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:40.948848Z","title":"Taming transformers for high-resolution image synthesis","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.948848Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:045f2cafeb751bd904365dff481e040c6f2b91ffe0dc47e9a4b96a6432aebe50","observation_id":"2a939be7-30af-4e14-bb95-8efc9d34d861","resolution":{"observed_at":"2026-08-06T17:59:40.948848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:41.025078Z","title":"Eva: Exploring the limits of masked visual representa- tion learning at scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.025078Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:b15c8e4671d792fec60116343a48131237cb280f2262a3f96bb7a08035400154","observation_id":"9cd767cd-5111-4302-83f8-a573f826ccab","resolution":{"observed_at":"2026-08-06T17:59:41.025078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00752","last_updated":"2024-05-31T17:55:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-01T18:01:34Z","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00752","snapshot_observed_at":"2026-08-06T17:59:41.141507Z","title":"Mamba: Linear-time sequence modeling with selective state spaces","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.141507Z"},"links":{"cited_paper":"/paper/2312.00752","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:fed64eefef62098afa1a1cde2e8cc135339cebd6de0a891b70ca292d02b6d364","observation_id":"faa02641-e8c6-4512-b4f3-9c82e76349ab","resolution":{"observed_at":"2026-08-06T17:59:41.141507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.880213Z","title":"Star-transformer","venue":null,"work_id":"e9de46e6-2f57-4240-a037-f6217fae3267","year":2019},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.197247Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:c2a6bea47f64541203b5e5e41b63f52135db591a91e8436431a1dfacaf787f54","observation_id":"57937788-8758-4a72-a04e-e97bf210e2da","resolution":{"observed_at":"2026-08-06T17:59:50.946490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.708601Z","title":"Girshick","venue":null,"work_id":"447c3316-ace0-40a6-9004-c227172e593d","year":2019},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.326298Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:12850153b911c5aba8bef4195700b65caa1cdcc2710849fefd09a5d9c196e3aa","observation_id":"6bc10f52-6175-4371-a10a-c8ffc204c68f","resolution":{"observed_at":"2026-08-06T17:59:50.779479Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.554125Z","title":"Flatten transformer: Vision transformer using focused linear attention","venue":null,"work_id":"de008ce0-95ba-4f5b-ac2a-fa405ae325fd","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.414485Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:b67c06a5078af7d47254dc0090b973f14b47bec671fdf466aa0dd5ec4c8f20f4","observation_id":"23044086-259f-4648-97ec-2fe855688be0","resolution":{"observed_at":"2026-08-06T17:59:50.601022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:41.495681Z","title":"Deep residual learning for image recognition","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.495681Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:e9f48fd872cbce9201c56156e24d1a0d93295b679f3e6223d36faadb3c2edf06","observation_id":"0c8ae67b-9f9f-4a9a-943c-2feadba750c2","resolution":{"observed_at":"2026-08-06T17:59:41.495681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.365951Z","title":"Mask r-cnn","venue":null,"work_id":"5c188f96-25b0-41d7-bf14-3c6f5d40003e","year":2017},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.618278Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:064b9b8fb18334b1002a73c72b85a845d7080527070d709ca227ca274609aee0","observation_id":"f48e18d7-fa88-41be-be1d-a6504c5142d2","resolution":{"observed_at":"2026-08-06T17:59:50.460850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.212780Z","title":"Interformer: Real-time interactive image segmentation","venue":null,"work_id":"3db5db1c-8257-4a4b-981f-bdcc27f15ee7","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.742838Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:2269404205f96600fd9035dad5071c4709ec38cd4b3d2b46de2259e8a3c7d52c","observation_id":"18b83410-5428-4233-b66d-4c4ca993df9e","resolution":{"observed_at":"2026-08-06T17:59:50.293455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.02109","last_updated":"2024-11-23T01:44:00Z","snapshot_observed_at":"2026-07-06T18:40:10.103133Z","submitted_at":"2024-07-02T09:51:56Z","title":"HRSAM: Efficient Interactive Segmentation in High-Resolution Images","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.02109","snapshot_observed_at":"2026-08-06T17:59:41.859791Z","title":"Hrsam: Efficiently seg- ment anything in high-resolution images","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.859791Z"},"links":{"cited_paper":"/paper/2407.02109","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:05ba3ddef0484cf196e80f945ad0aa0b5a7a1c9dbf3bfdf78b4741c716f0db54","observation_id":"5e659fb3-e108-49b7-bc20-9d25b4b55524","resolution":{"observed_at":"2026-08-06T17:59:41.859791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:50.052272Z","title":"Interactive image seg- mentation via backpropagating refinement scheme","venue":null,"work_id":"8c38c54d-3da9-4e30-a6c6-352f980a138e","year":2019},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:41.979461Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:5cd2e94d3a51cf75f0dad730062ad7bfd1b999c483915b6124615ed14db7de74","observation_id":"039b08ca-caff-40c4-b4f9-45d403c82f10","resolution":{"observed_at":"2026-08-06T17:59:50.105380Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:49.855488Z","title":"Segment anything in high quality","venue":null,"work_id":"92972ca8-a965-4649-9161-48f95c3a8055","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.093943Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:43a1e81ff73a2566ce45419830cf5a0137a55750bc55dc583d46fe62b4d6976b","observation_id":"13a2bf29-6b91-4c73-b32b-0be3d0d4d076","resolution":{"observed_at":"2026-08-06T17:59:49.948278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:49.702041Z","title":"Transformers in vision: A survey","venue":null,"work_id":"68f527c9-3bdb-464d-8939-2a06eb06590c","year":2022},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.160848Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:8dde5701a2884133bc425ffe4c92badec58d864cd2f7896582167e3e66fa1538","observation_id":"721247d7-4a60-41e1-9626-66566a199402","resolution":{"observed_at":"2026-08-06T17:59:49.779183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:49.516526Z","title":"Segment any- thing","venue":null,"work_id":"14fc30ff-5a71-43db-93ff-4130f0e2c564","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.279968Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:d70dd44e4201efbbf6b95c0aab62918ded658a7cee732cd66222b3e2028ae268","observation_id":"510cd5c1-3fb1-4a8b-a843-6cc3ccbcebbc","resolution":{"observed_at":"2026-08-06T17:59:49.583175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:49.196204Z","title":null,"venue":null,"work_id":"81685265-6d19-4b8c-98db-f4113bbf14da","year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.391399Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:6ee8844d55ad2257f2ae9295c66a040ad7a4b65ecbd0c658d2079c657f9419e8","observation_id":"e2bdc49e-006c-4078-8871-b0af4fae5d64","resolution":{"observed_at":"2026-08-06T17:59:49.360698Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.00692","last_updated":"2024-05-01T05:10:13Z","snapshot_observed_at":"2026-08-06T07:43:56.889679Z","submitted_at":"2023-08-01T17:50:17Z","title":"LISA: Reasoning Segmentation via Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.00692","snapshot_observed_at":"2026-08-06T17:59:42.435568Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.435568Z"},"links":{"cited_paper":"/paper/2308.00692","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:c9b75a1b0f92d750dec8f8627989ecb6cc73310d14fedcd3a8be029c3c0de770","observation_id":"e1982561-f511-4d16-a33c-2e135b68235f","resolution":{"observed_at":"2026-08-06T17:59:42.435568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:48.738371Z","title":"Exploring plain vision transformer backbones for object de- tection","venue":null,"work_id":"c4d01c60-c2f0-4955-9d9c-c06bd052d024","year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.488811Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:0d693dc988a7591515a87d60a2af4244dd4b1fcebb088052639a708da308bb54","observation_id":"3ea5ba65-c0c8-4629-ac84-46abbcd06e0b","resolution":{"observed_at":"2026-08-06T17:59:48.980421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:48.576966Z","title":"Interactive image segmentation with latent diversity","venue":null,"work_id":"2062dccb-a689-41b5-945c-7c6f3ebc3988","year":2018},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.557505Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:d07b304f2a3ae22f80acef139df01896b4a9e2f0dd1516ce81d227f2ff16bf19","observation_id":"fb3b5c69-3955-4d6a-9e84-1d6c548bab83","resolution":{"observed_at":"2026-08-06T17:59:48.618076Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:48.202618Z","title":"Lawrence Zitnick","venue":null,"work_id":"b65ae09d-15e5-4bc6-909d-156401df45e3","year":2014},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.727012Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:3f98974c21a619230fec09f1f6ef27eae687df6f918e1e9e6654342b2fd0f00b","observation_id":"63f3bcea-9b9b-4829-b9ff-37fc53304fd3","resolution":{"observed_at":"2026-08-06T17:59:48.323770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:47.925432Z","title":"Interactive image segmentation with first click attention","venue":null,"work_id":"759d075e-255f-48be-b4ee-c20194009a77","year":2020},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.789996Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:598d0e3d0dbd418decb9913c6096179ce28ac2e242df979ed2edb2c69bc4d805","observation_id":"c452a0a2-fb5c-4b5d-b66f-39c214892251","resolution":{"observed_at":"2026-08-06T17:59:48.043298Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16354","last_updated":"2024-02-25T09:48:53Z","snapshot_observed_at":"2026-07-06T16:24:53.814969Z","submitted_at":"2023-09-28T11:26:52Z","title":"Transformer-VQ: Linear-Time Transformers via Vector Quantization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16354","snapshot_observed_at":"2026-08-06T17:59:42.865077Z","title":"Transformer-vq: Linear-time transformers via vector quantization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.865077Z"},"links":{"cited_paper":"/paper/2309.16354","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:04c2829b4fdf14aa33e72ce157e9bd70ea9aa6c9d2ccf7f310e796dc27a60124","observation_id":"ca0054aa-37ef-4ed8-aa3c-6908b3dafa19","resolution":{"observed_at":"2026-08-06T17:59:42.865077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T17:59:42.964375Z","title":"Deepseek-v3 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.964375Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:3ea48509f59c219f0cf1535218d1243f0cff9c3dc97d8146a67942b40ca5900a","observation_id":"329fe068-ed4e-49ae-989b-135a43cbe3a6","resolution":{"observed_at":"2026-08-06T17:59:42.964375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.11006","last_updated":"2023-03-11T19:36:34Z","snapshot_observed_at":"2026-08-07T15:15:55.287442Z","submitted_at":"2022-10-20T04:20:48Z","title":"SimpleClick: Interactive Image Segmentation with Simple Vision Transformers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.11006","snapshot_observed_at":"2026-08-06T17:59:43.065723Z","title":"Simpleclick: Interactive image segmentation with sim- ple vision transformers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.065723Z"},"links":{"cited_paper":"/paper/2210.11006","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:526e67ff17cfbde225550d2fb5f9581e16f96a2fd68289de9f0663b3aad72070","observation_id":"71ee2ea2-52b4-4cea-ba5f-8a4c8bd46526","resolution":{"observed_at":"2026-08-06T17:59:43.065723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:47.684181Z","title":"Pseudoclick: Interactive image segmentation with click imi- tation","venue":null,"work_id":"3872b20a-1326-49e9-a065-d9620f0a6e5b","year":2022},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.122967Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:9f7b700efa9fa10c8f26d00359e0b348c838d7b1609000f9a63d4a84f2db6d68","observation_id":"0a4c9921-87d5-4a7a-b679-54aca9b73b6a","resolution":{"observed_at":"2026-08-06T17:59:47.805495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.00741","last_updated":"2024-03-31T17:02:24Z","snapshot_observed_at":"2026-07-06T17:53:44.044640Z","submitted_at":"2024-03-31T17:02:24Z","title":"Rethinking Interactive Image Segmentation with Low Latency, High Quality, and Diverse Prompts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.00741","snapshot_observed_at":"2026-08-06T17:59:43.181168Z","title":"Rethinking interactive image segmentation with low latency, high quality, and diverse prompts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.181168Z"},"links":{"cited_paper":"/paper/2404.00741","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:3b8ad2aec4870a666ddf7d61a248d6f45ba929986b1fd193c9bb61851fd1bb0a","observation_id":"ffa45373-8032-4bee-922d-e432b9bc3a64","resolution":{"observed_at":"2026-08-06T17:59:43.181168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:43.255738Z","title":"Swin transformer: Hierarchical vision transformer using shifted windows","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.255738Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:b7f1f78e5a24c704764fba8c61b00a59ee5cb9a13bc7553eb05417720286283e","observation_id":"b93184ae-072a-40f8-9aca-da083c05f840","resolution":{"observed_at":"2026-08-06T17:59:43.255738Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.12306","last_updated":"2024-04-01T16:18:16Z","snapshot_observed_at":"2026-07-06T15:19:28.033750Z","submitted_at":"2023-04-24T17:56:12Z","title":"Segment Anything in Medical Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.12306","snapshot_observed_at":"2026-08-06T17:59:43.355075Z","title":"Segment anything in medical images","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.355075Z"},"links":{"cited_paper":"/paper/2304.12306","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:f94c60d35b71367cf221e6dfb2182e127ada6a845f2cad3d2a3cffdbecbf31bf","observation_id":"7ae41f4b-92fe-4efb-b2fe-640468955fb8","resolution":{"observed_at":"2026-08-06T17:59:43.355075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:47.341509Z","title":"Deep extreme cut: From extreme points to object segmentation","venue":null,"work_id":"c5b2a1db-6141-48de-8d8c-4735ee280cd5","year":2018},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.451734Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:f42f4af3dad104ef87489e1a5d3d2647a9fb4d7d9536ea9efb48cde3dc874db4","observation_id":"73251273-ef52-4c90-b8e4-ba5c479e7bf0","resolution":{"observed_at":"2026-08-06T17:59:47.500614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:43.527111Z","title":"Segment anything model for medical image analysis: an experimental study","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.527111Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:ec113848216f9063653f9216e40be56349e15168851620ca09fbea431b691317","observation_id":"0443b593-47dc-4ad8-8ce5-8aae21380599","resolution":{"observed_at":"2026-08-06T17:59:43.527111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.15505","last_updated":"2023-10-12T07:55:05Z","snapshot_observed_at":"2026-07-06T16:24:17.829828Z","submitted_at":"2023-09-27T09:13:40Z","title":"Finite Scalar Quantization: VQ-VAE Made Simple","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.15505","snapshot_observed_at":"2026-08-06T17:59:43.608657Z","title":"Finite scalar quantization: Vq-vae made simple","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.608657Z"},"links":{"cited_paper":"/paper/2309.15505","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:fc54558005fa3f05d0f73960ad3d5434e4a2b5c4a216fa420f6ebcc2b92932f0","observation_id":"b035f61b-7131-4304-81e5-9159d52efd71","resolution":{"observed_at":"2026-08-06T17:59:43.608657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:47.055106Z","title":"Gross, and Alexander Sorkine- Hornung","venue":null,"work_id":"ae898552-9830-4e24-8cdf-f00e33b20f62","year":2016},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.681941Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:00fec625e1665e79bd8763f8ebca2395e47172d5db814bc7dd5faf06ae56e028","observation_id":"8ed33318-551b-4119-b464-9482ebe37a36","resolution":{"observed_at":"2026-08-06T17:59:47.187738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.05682","last_updated":"2022-10-10T16:36:47Z","snapshot_observed_at":"2026-08-06T17:49:01.053607Z","submitted_at":"2021-12-10T17:25:07Z","title":"Self-attention Does Not Need $O(n^2)$ Memory","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.05682","snapshot_observed_at":"2026-08-06T17:59:43.755282Z","title":"Rabe and Charles Staats","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.755282Z"},"links":{"cited_paper":"/paper/2112.05682","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:037c0d5127b965aa448dd9962286ff257b832090c0c11d6b11d9cf53733d01b6","observation_id":"38941a85-0e27-44a0-bfcd-2a070ef9bc96","resolution":{"observed_at":"2026-08-06T17:59:43.755282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:46.688494Z","title":"Petrov, Olga Barinova, and Anton Konushin","venue":null,"work_id":"16168a43-5f7b-4e0b-80dd-2c9033e50460","year":2020},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.827520Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:2a156ecfe38a9da0b9e1e2a6a4170b658c418696350838760711ce60aed50411","observation_id":"a0001edd-c4e3-402b-aed2-3735caf3e73c","resolution":{"observed_at":"2026-08-06T17:59:46.839249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:43.915454Z","title":"Roformer: Enhanced transformer with rotary position embedding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.915454Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:69d3690e594a359578f77bf0c28c2cb546da206194b6f29863254ea79bc4bb4b","observation_id":"6fc9de30-9d6a-4270-b772-4fe39c8996b9","resolution":{"observed_at":"2026-08-06T17:59:43.915454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:43.974606Z","title":"Neural discrete representation learning","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:43.974606Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:3f75db6dce116b2e8581111400e84a5169117e5379918417ed154d581966fd8b","observation_id":"97a92bc1-5798-4a02-95f0-859d027b5316","resolution":{"observed_at":"2026-08-06T17:59:43.974606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:46.377093Z","title":"Gomez, Lukasz Kaiser, and Illia Polosukhin","venue":null,"work_id":"4366fea4-8dd9-416f-9e56-33e7c9b53575","year":2017},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.042802Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:c4a8d58edcdcc5fbaef17529f99245342085b7cefb0608db7c6e0e5836baef1c","observation_id":"7adda1d5-3e5b-4f11-be44-cc5cc766deca","resolution":{"observed_at":"2026-08-06T17:59:46.517556Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12214","last_updated":"2025-02-06T22:16:59Z","snapshot_observed_at":"2026-08-06T15:36:55.343925Z","submitted_at":"2024-10-16T04:19:28Z","title":"Order-aware Interactive Segmentation","version":3},"cited_work":{"arxiv_id":"2410.12214","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.12214","snapshot_observed_at":"2026-08-06T17:59:45.128674Z","title":"Order-aware Interactive Segmentation","venue":"cs.CV","work_id":"b5a378bf-8c7b-4183-ab5c-cb1c28927e22","year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.103291Z"},"links":{"cited_paper":"/paper/2410.12214","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:bdd72e0fb19439b9c0593a6451ce18b03dc0dc8c76b36eded6e897799527e4c7","observation_id":"e37945da-c806-43af-9c51-5b8ecaafa336","resolution":{"observed_at":"2026-08-06T17:59:45.174724Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:46.061879Z","title":"Internimage: Exploring large-scale vi- sion foundation models with deformable convolutions","venue":null,"work_id":"39c5fc08-e6c4-41eb-a646-174eb100db9b","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.162339Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:33fc6aae9c8572efbc1a2af5aed36ee792aa749d21c62b8ede070a60a715ef32","observation_id":"b83b033a-e8a4-46d1-85a2-ecaef160dbff","resolution":{"observed_at":"2026-08-06T17:59:46.188030Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00863","last_updated":"2023-12-01T18:31:00Z","snapshot_observed_at":"2026-07-06T16:55:49.813116Z","submitted_at":"2023-12-01T18:31:00Z","title":"EfficientSAM: Leveraged Masked Image Pretraining for Efficient Segment Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00863","snapshot_observed_at":"2026-08-06T17:59:44.221558Z","title":"Efficientsam: Leveraged masked image pretraining for efficient segment anything","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.221558Z"},"links":{"cited_paper":"/paper/2312.00863","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:7cf12fd2519f95022c358b6e746cb17b13795754d64b6024fb4cb057a78623e5","observation_id":"ee6d4960-afe5-4b1c-9271-2392a7a93bad","resolution":{"observed_at":"2026-08-06T17:59:44.221558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04009","last_updated":"2024-05-07T04:57:25Z","snapshot_observed_at":"2026-08-06T00:06:25.546349Z","submitted_at":"2024-05-07T04:57:25Z","title":"Structured Click Control in Transformer-based Interactive Segmentation","version":1},"cited_work":{"arxiv_id":"2405.04009","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.04009","snapshot_observed_at":"2026-08-06T17:59:44.942607Z","title":"Structured Click Control in Transformer-based Interactive Segmentation","venue":"cs.CV","work_id":"a6809a26-d379-4ae3-b5c8-8cb5af31d486","year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.282657Z"},"links":{"cited_paper":"/paper/2405.04009","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:b062ce5c1414ffa5f6deb49edafde93ffb8891e086a86f775e4affc58c9cc4c0","observation_id":"db734360-93ec-4cb0-be7c-86b44932a057","resolution":{"observed_at":"2026-08-06T17:59:45.006158Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:45.767750Z","title":"Price, Scott Cohen, Jimei Yang, and Thomas S","venue":null,"work_id":"e18dfc9c-f744-4ef6-a79b-2d8be1e12e05","year":2016},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.339765Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:89da480e54a14b4788493156a8d7f3ba8030ae6829144cf1005011d987db5359","observation_id":"c73dd96e-6bd5-4651-bb50-1ddb747eb924","resolution":{"observed_at":"2026-08-06T17:59:45.876912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.04627","last_updated":"2022-06-05T01:57:58Z","snapshot_observed_at":"2026-07-06T11:56:10.689708Z","submitted_at":"2021-10-09T18:36:00Z","title":"Vector-quantized Image Modeling with Improved VQGAN","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.04627","snapshot_observed_at":"2026-08-06T17:59:44.400662Z","title":"Vector-quantized image modeling with improved vqgan","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.400662Z"},"links":{"cited_paper":"/paper/2110.04627","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:9107e543d0dc3345563e5255a5fed6281ad52bbd6c2e74b1d3b8174cf49a6e7c","observation_id":"7ff96d30-366d-477c-b2cf-5c28cf7f9484","resolution":{"observed_at":"2026-08-06T17:59:44.400662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14289","last_updated":"2023-07-01T07:26:22Z","snapshot_observed_at":"2026-07-06T15:46:29.060519Z","submitted_at":"2023-06-25T16:37:25Z","title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14289","snapshot_observed_at":"2026-08-06T17:59:44.470606Z","title":"Faster segment anything: Towards lightweight sam for mo- bile applications","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.470606Z"},"links":{"cited_paper":"/paper/2306.14289","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:3d2d1a0532fd6d7cdb3fb0fb0689079dac3f42cf2ed3123beac8e7cccecf2bfa","observation_id":"a07228c9-062e-49b5-9d5a-54ab5f4ecd8f","resolution":{"observed_at":"2026-08-06T17:59:44.470606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19423","last_updated":"2024-02-29T18:22:12Z","snapshot_observed_at":"2026-08-03T19:57:28.408015Z","submitted_at":"2024-02-29T18:22:12Z","title":"Leveraging AI Predicted and Expert Revised Annotations in Interactive Segmentation: Continual Tuning or Full Training?","version":1},"cited_work":{"arxiv_id":"2402.19423","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.19423","snapshot_observed_at":"2026-08-06T17:59:44.785740Z","title":"Leveraging AI Predicted and Expert Revised Annotations in Interactive Segmentation: Continual Tuning or Full Training?","venue":"cs.CV","work_id":"3d962c4e-fb25-4df6-b4c4-2def82286471","year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.501671Z"},"links":{"cited_paper":"/paper/2402.19423","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:259329e2b4c3fa94d3670c0eb8a30ccf25e2a7cd4e8583bbf32508c82dfb13fc","observation_id":"52ab7135-a609-4e82-8ef2-6c08c56889d0","resolution":{"observed_at":"2026-08-06T17:59:44.831422Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07548","last_updated":"2024-06-11T17:59:53Z","snapshot_observed_at":"2026-08-05T21:17:06.014910Z","submitted_at":"2024-06-11T17:59:53Z","title":"Image and Video Tokenization with Binary Spherical Quantization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07548","snapshot_observed_at":"2026-08-06T17:59:44.578141Z","title":"Image and video tokenization with binary spherical quantization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.578141Z"},"links":{"cited_paper":"/paper/2406.07548","citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:b20571a366c91fc9206bdf60a22b566982d444c5248f54a8ae06d456b90fc697","observation_id":"6be68bb3-d653-4d4c-832b-eaf9de91d993","resolution":{"observed_at":"2026-08-06T17:59:44.578141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:45.499927Z","title":"Online clustered code- book","venue":null,"work_id":"b477f51e-6651-4d7a-9df4-1ad46ec77426","year":2023},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:44.658585Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:8d86c8bce410016e20c14b8676a146ae330e4772f87d46957686cecf9d3772ad","observation_id":"25e12b38-6e27-4056-ba6f-6a6f91a887d8","resolution":{"observed_at":"2026-08-06T17:59:45.623460Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:48.420358Z","title":null,"venue":null,"work_id":"b5f595d5-2914-41e1-b3bf-cc819b6d2a7b","year":2018},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":585,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:42.632091Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:2ff69924fb7d03b4fa1f4859f161f7dca0389f66cc912d847759adcb30d2730b","observation_id":"4376c965-97a6-4055-a413-c858a5cfd599","resolution":{"observed_at":"2026-08-06T17:59:48.501036Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:59:51.057425Z","title":null,"venue":null,"work_id":"19dce115-8719-447d-8c6b-3472a6f84046","year":2021},"citing_paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:40.757928Z"},"links":{"citing_paper":"/paper/2507.09612"},"observation_digest":"sha256:21046481d17efd64f134bbf9b7539b19d67248e3afb442f0d88b1134a004b294","observation_id":"886b1b02-97bc-4557-b83c-ed82e9b33a50","resolution":{"observed_at":"2026-08-06T17:59:51.130413Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.09612","last_updated":"2025-07-13T12:33:37Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T05:37:36.169460Z","submitted_at":"2025-07-13T12:33:37Z","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive"},"reference_resolution":{"displayed":57,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":1,"unresolved":26,"verified_exact":3,"verified_fuzzy":26},"total_outbound_references":57},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 57 of 57 outbound references and 0 inbound Pith citation observations for arXiv:2507.09612."}