{"as_of":"2026-08-10T14:55:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a1030a7c348cef86f9b692eaaf4af35a87b0732ced2c38773632030eeee126b2","coverage":[{"denominator":68,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":68,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:23:15.210670Z","state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.05798/citation-record","integrity":"/paper/2507.05798/integrity","json":"/paper/2507.05798/citation-record.json","paper":"/paper/2507.05798"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2211.01324","last_updated":"2023-03-14T00:22:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-11-02T17:43:04Z","title":"eDiff-I: Text-to-Image Diffusion Models with an Ensemble of Expert Denoisers","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.01324","snapshot_observed_at":"2026-08-06T19:23:13.558538Z","title":"ediff-i: Text-to-image diffusion models with an ensem- ble of expert denoisers","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:13.558538Z"},"links":{"cited_paper":"/paper/2211.01324","citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:343157f76093db1ecf4b8b05abadb3126d59c4d3dc2873af6cacb6eb912d0ca2","observation_id":"e6c1f154-424c-4e6e-8c97-0885f4d72d77","resolution":{"observed_at":"2026-08-06T19:23:13.558538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.433751Z","title":"Hico: A benchmark for recognizing human-object interactions in images","venue":null,"work_id":"fcc197fe-5a9b-4d42-98f7-a467c2dffc06","year":null},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:13.604400Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:56e22b37d40400dcaec3c2103b8b068e2bb99f094b2dee4d3be5e36897926ca2","observation_id":"a5cbe5f6-98ba-4344-ab0f-838db8bbb4c1","resolution":{"observed_at":"2026-08-06T19:23:16.438037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.420476Z","title":"Spatialvlm: Endow- ing vision-language models with spatial reasoning capabili- ties","venue":null,"work_id":"16b6c266-e0b7-461d-81d0-dff0e7dcd8d1","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:13.778985Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:0bd18c782e020c01ca54ea0a502e8974f472a6ee79602ea849c0aad0b72dde41","observation_id":"005633a3-85e2-4f05-9326-f3d7e62cff7a","resolution":{"observed_at":"2026-08-06T19:23:16.424545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.407411Z","title":"Expanding scene graph boundaries: fully open-vocabulary scene graph generation via visual-concept alignment and retention","venue":null,"work_id":"b5a5a840-91a0-45a6-8f14-8540fa37c2d8","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:13.901264Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:e27fba35c366ac11f13fc8cc82bc35f4d132f0030a77ea1eeeabdd52b9516a27","observation_id":"1be1c4b7-8b93-4eb6-8d1f-1496ce0c38d8","resolution":{"observed_at":"2026-08-06T19:23:16.411541Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.394191Z","title":"Spatial- rgpt: Grounded spatial reasoning in vision-language mod- els","venue":null,"work_id":"90f23ca9-7204-47ea-ba22-73a144a7f027","year":2025},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.000753Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:9b40a80617893aca277197b2583ab2fc0713bd7fd509aeca99a1504bd410aa1a","observation_id":"fcc6fb54-c3fd-4be6-8248-ccb32eaadad3","resolution":{"observed_at":"2026-08-06T19:23:16.398275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.380731Z","title":"Masked-attention mask transformer for universal image segmentation","venue":null,"work_id":"f0e82e78-d8c7-470e-879e-158ae1acf4c4","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.118018Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:7a50f5608311f25a368d5c65b2ccdd9e4e5a462dcf50f5f9804c4924cc9e8abb","observation_id":"4eb96ac1-1921-4d84-9f32-c7d91bdfb551","resolution":{"observed_at":"2026-08-06T19:23:16.384940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.367133Z","title":"Recovering the unbiased scene graphs from the biased ones","venue":null,"work_id":"fdb6820d-32c7-4f6b-a6fe-ea2de5a0de6a","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.227404Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:89574922fdb8fd336c213d38ba9b939775d6f6c97a6dad105aa759dcdbfceab5","observation_id":"dcd6c2bc-d51c-4c24-9184-d58110c75fd2","resolution":{"observed_at":"2026-08-06T19:23:16.371152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.353589Z","title":"Reltr: Relation transformer for scene graph generation.IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(9):11169–11183, 2023","venue":null,"work_id":"36d841bf-72f7-4779-a9bc-159640a635d8","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.387287Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:0c7cdf28625917127c0a24d5a052240181d747cef8aede38d52322d2ebf21884","observation_id":"15167482-b84b-4d8e-85d4-87a7163c1d0b","resolution":{"observed_at":"2026-08-06T19:23:16.358031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.339685Z","title":"De- coupling zero-shot semantic segmentation","venue":null,"work_id":"9df1f48f-a0fe-4758-8a46-8f25848fe30d","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.489106Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:f55001a52b14fb7676d2e86dc2bd752c8efc7298e43dca73c147fb9bf5aa4e8e","observation_id":"170c9556-717f-43f7-9b4a-3e42ebe5f63b","resolution":{"observed_at":"2026-08-06T19:23:16.344306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.326359Z","title":"Concept sliders: Lora adaptors for precise control in diffusion models","venue":null,"work_id":"9e21fb2e-6b96-470a-b338-47574e0ed503","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.635239Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:cec27f15be8e2921ef3e039e071893ececd369f9d15c478223757ab7639e5464","observation_id":"6ee591b2-59ea-4bf1-b39c-3338f986638d","resolution":{"observed_at":"2026-08-06T19:23:16.330702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.312049Z","title":"ican: Instance- centric attention network for human-object interaction detec- tion","venue":null,"work_id":"c86faf57-2fea-447b-bb7f-e91b828aef4b","year":2018},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.761753Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:b7935b64d29d7a440bac1d67c6428dde5ea55d9c9fd07d429a8ae11da1e0cd2c","observation_id":"0844e64d-0997-4aee-84f6-eafa729a12c5","resolution":{"observed_at":"2026-08-06T19:23:16.316366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.298814Z","title":"Open-vocabulary object detection via vision and language knowledge distillation","venue":null,"work_id":"067a1916-369e-48f1-baf0-3d73cf6d7141","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.844366Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:761251fd9d4eab90a05f151e67b80f06f5d45de2528f4cd74ce91a9946e2ca5a","observation_id":"38bedc89-6f44-488e-9072-617542aaa39a","resolution":{"observed_at":"2026-08-06T19:23:16.302822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.285092Z","title":"Dsgg: Dense relation transformer for an end-to-end scene graph generation","venue":null,"work_id":"a3d53948-9bca-4ba1-91d9-89798deba29c","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.973198Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:221cf3d9981ff21143600eb38b37ac7a354d32c108a6008d4eb0eabee1e32348","observation_id":"4a89acf0-82f2-42ad-bbd9-b33ea4c630ae","resolution":{"observed_at":"2026-08-06T19:23:16.289314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.271013Z","title":"Learning from the scene and borrowing from the rich: tackling the long tail in scene graph generation","venue":null,"work_id":"87e11dc0-a090-4cf5-a0d0-057698a67952","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.977613Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:f0068e6997f56cb03aacc07839f35df21345e667d3da3cf49addcadd0577be73","observation_id":"ce04be5c-e975-4890-9ee2-b5cd047c104a","resolution":{"observed_at":"2026-08-06T19:23:16.275295Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.08600","last_updated":"2021-08-19T10:13:55Z","snapshot_observed_at":"2026-07-06T11:39:34.310174Z","submitted_at":"2021-08-19T10:13:55Z","title":"Semantic Compositional Learning for Low-shot Scene Graph Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.08600","snapshot_observed_at":"2026-08-06T19:23:14.982119Z","title":"Semantic compositional learning for low-shot scene graph generation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.982119Z"},"links":{"cited_paper":"/paper/2108.08600","citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:67938d907658173357020a8a3ac0629d0a5ab08c50d70e358101fbbae8221c1d","observation_id":"d319bbd3-3677-4553-bfeb-b1b40f6c5861","resolution":{"observed_at":"2026-08-06T19:23:14.982119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.256401Z","title":"Ex- ploiting scene graphs for human-object interaction detection","venue":null,"work_id":"526afcd1-228f-417a-a9bc-a6ff01d437b2","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.986482Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:1d04993328902c1407978e765f34b8e2c0f9063ddd49b0d2cf628fd95df2eaae","observation_id":"df8ae0da-2210-4c84-bacf-2e746426e7f9","resolution":{"observed_at":"2026-08-06T19:23:16.260984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.242241Z","title":"To- wards open-vocabulary scene graph generation with prompt- based finetuning","venue":null,"work_id":"76b7e5d9-f66a-4945-b313-d96f10c5b44b","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.990928Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:01ca277cdeda635715f22b01ee24935a569b69e2ffed5ea83229e92a0159ad4e","observation_id":"5d1dcdcb-0adf-417e-9c2f-a059f2ca0c83","resolution":{"observed_at":"2026-08-06T19:23:16.246648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.226860Z","title":"To- ward a unified transformer-based framework for scene graph generation and human-object interaction detection","venue":null,"work_id":"70fc2203-8c75-4006-a0e1-022afe39d84a","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.994859Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:0f73a59901e493e1232d381310abed9f7e19b6c66a09ed545a3263b7e42f16eb","observation_id":"b82d42b7-4d1e-43f6-a206-0a28444edef6","resolution":{"observed_at":"2026-08-06T19:23:16.231947Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.212419Z","title":"Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen- Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen","venue":null,"work_id":"07384bb2-8769-4919-814b-46f2e64ea621","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:14.998881Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:6c69825a72ab069ed25c147f8f2f4bd6ddc6dd8c7f3d66630267512c57ba031e","observation_id":"06435c4a-dfbd-4a64-b90e-c4636381af1a","resolution":{"observed_at":"2026-08-06T19:23:16.216462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.196925Z","title":"Egtr: Extracting graph from trans- former for scene graph generation","venue":null,"work_id":"90a7225b-495a-49fe-bbfd-7b13ae73006b","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.003205Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:0e34994638f78e6db3d15a62a3952b4b6ce9878ebaba50b687d0b3827385220c","observation_id":"da90bef4-3968-4ad9-87c7-bd06190c5f16","resolution":{"observed_at":"2026-08-06T19:23:16.201664Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.181531Z","title":"Zero-shot scene graph relation prediction through commonsense knowledge inte- gration","venue":null,"work_id":"7d8c4d47-6956-4e2c-ab54-e0472507fcc1","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.007201Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:0235110a0216e0df4f770e1951e00ba3d1cccee288b52a41abc3c3e1668a6e7d","observation_id":"37443cb2-411b-4f47-ad8f-a629003eb7e4","resolution":{"observed_at":"2026-08-06T19:23:16.185701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.167096Z","title":"Imagic: Text-based real image editing with diffusion models","venue":null,"work_id":"eb22562a-11f3-4168-aec0-0873cac5bd78","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.011704Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:0ff2d9f31b9d14e6efba5cf4d25a85449d7c81c34694a526db13f4e3453933ac","observation_id":"9dd773e6-a11b-4497-bf4a-7a5cf6b03f21","resolution":{"observed_at":"2026-08-06T19:23:16.171509Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.152392Z","title":"Text-image align- ment for diffusion-based perception","venue":null,"work_id":"e66a3070-a2df-4624-bc1e-a97112ad2e0f","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.015861Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:1f3e15d99d76bbed15e8b03eb8e8b3afb6647c4e2f03688b24e7136b4a476be8","observation_id":"68dc0939-d405-42e9-8f93-6b34c55aace4","resolution":{"observed_at":"2026-08-06T19:23:16.156666Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.138742Z","title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations","venue":null,"work_id":"022e8fe4-4ef4-4e72-b723-d3711b81accd","year":2017},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.019691Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:1174562cf53e2b06668667b4944d660ca0b7e4a11672376e97bf7954a367b3bc","observation_id":"be956cd2-2ca7-478d-b0fc-fc2c0d07aadc","resolution":{"observed_at":"2026-08-06T19:23:16.143198Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.123778Z","title":"Topviewrs: Vision-language models as top-view spatial reasoners","venue":null,"work_id":"9239f237-efd8-41a5-ba13-792da251cb5f","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.024093Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:5269ccf31b5c2b0708b8be8efd74968dc0a2a1e4adcec4f64848f2d2f877d0b6","observation_id":"fe5635cc-04e6-4064-8929-226ff622c790","resolution":{"observed_at":"2026-08-06T19:23:16.128466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.028859Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.028859Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:bf81ea7a40ab3d2d6784c79257a6d4cf7a372a906d6a9ec179b85be0cb417e07","observation_id":"f046e346-5bd2-492f-a536-fe5dc20f3de0","resolution":{"observed_at":"2026-08-06T19:23:15.028859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.099181Z","title":"Panoptic scene graph genera- tion with semantics-prototype learning","venue":null,"work_id":"19d93e1f-86bb-4969-baf6-464f89b014a4","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.033359Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:010508890aa3647e785862ed77f324ecac7e384bafd04654b6d3a5d86602baf4","observation_id":"eae83f39-788f-492c-8ddc-98fabd4a1e67","resolution":{"observed_at":"2026-08-06T19:23:16.103340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.042930Z","title":"Grounded language-image pre-training","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.042930Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:77f61894b49832931880c175af2a15c9a35834159f1bf71233475263d4799db7","observation_id":"0ffe0184-9e90-4a6e-a8ee-f0e8476d4c99","resolution":{"observed_at":"2026-08-06T19:23:15.042930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.074952Z","title":"Sgtr: End- to-end scene graph generation with transformer","venue":null,"work_id":"603353b1-2ad1-4081-a178-fbd2d3ffca9f","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.047144Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:564184ffca39286b37a29ed914f2f0bb48e7fc5e0fcccc7a0735ea8e5ce6e1e9","observation_id":"1aac3644-1b55-4285-a068-f9eb0300f70c","resolution":{"observed_at":"2026-08-06T19:23:16.078982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.059601Z","title":"From pixels to graphs: Open-vocabulary scene graph generation with vision-language models","venue":null,"work_id":"43106136-fba5-4b3a-825a-0b875a6a387e","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.051619Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:2e11d18b46c13e8ed0d66e4f7973f32a0594980d30eea7c26c3abd910b97ea40","observation_id":"1ef41c09-9691-4db3-adff-23682b389116","resolution":{"observed_at":"2026-08-06T19:23:16.064671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.044279Z","title":"Open-vocabulary object segmentation with diffusion models","venue":null,"work_id":"b3b24f46-b963-4dba-87f6-33a1dedbb72f","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.056473Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:f64f3a5b06536ab1635dd8ab6607b73ff00e394e7e9f6f090e71de35e1e27f4e","observation_id":"8e1b958a-2675-4c82-a32e-ad75ea3a3ff0","resolution":{"observed_at":"2026-08-06T19:23:16.048713Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.029040Z","title":"Gps-net: Graph property sensing network for scene graph generation","venue":null,"work_id":"b9a46323-4c18-4232-a1fb-8aa41316681e","year":2020},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.061125Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:adb9e5dab97eaca41488eaf325abf4e8433ef5090cc4e6f566ce5d38a4939fe2","observation_id":"c6925a11-3925-4f35-ae00-d18804f114b3","resolution":{"observed_at":"2026-08-06T19:23:16.033751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:16.013696Z","title":"Path aggregation network for instance segmentation","venue":null,"work_id":"aa30b770-d6da-43c0-b14b-4a1e28a347bf","year":2018},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.065424Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:fa33c323298a102763f1fb5c0fa6cad2dd3d0abda9451562d4d436f07185f9e3","observation_id":"ce718288-c061-4752-8133-902602392a31","resolution":{"observed_at":"2026-08-06T19:23:16.018071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.997987Z","title":"Swin transformer: Hierarchical vision transformer using shifted windows","venue":null,"work_id":"adc31042-d8a8-4627-a7cc-f917f00de354","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.069551Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:73df2e9953d372003f8185b68efc2555f322c82266de1a6b2f690786f5698416","observation_id":"83209c0b-e7ef-4d13-9d91-7ab56c820be4","resolution":{"observed_at":"2026-08-06T19:23:16.002750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.982964Z","title":"Attending to graph transformers","venue":null,"work_id":"977c3e77-d437-4c63-8161-a48029f6375f","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.074156Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:9690b2dbfd00a1b9c9a3b5c1589713a7b123dbc2db3ddf1d0a068e6220e65d97","observation_id":"0a26314b-c230-407e-a187-3ede07ed7e14","resolution":{"observed_at":"2026-08-06T19:23:15.987147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.078590Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.078590Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:2ea20c43642b6dca545ffb511c765ce6612751a70427bdd9ce23d4e9e5e7c2d7","observation_id":"072be47a-50cb-41f5-add7-b45233eaab1a","resolution":{"observed_at":"2026-08-06T19:23:15.078590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.956508Z","title":"Faster r-cnn: Towards real-time object detection with region proposal networks","venue":null,"work_id":"25b36b71-5cd4-459a-96c8-cba32a291437","year":2015},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.082879Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:830d60d06b7f66a3bf74ae279d4ad3ae2eccfe7e3da5150296e00a6ae6ca880f","observation_id":"31b5836a-ae14-47c3-86f5-45832505cff6","resolution":{"observed_at":"2026-08-06T19:23:15.961046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.941952Z","title":"High-resolution image syn- thesis with latent diffusion models","venue":null,"work_id":"21edcea0-0661-40bc-b60e-17c283161c75","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.086847Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:814556c8db1c13985774f93d2a5d7e7d0facd0204fd214ea8cbe4e2cf32cddaf","observation_id":"14c277fa-c002-4bdd-9e0c-ab4717562e21","resolution":{"observed_at":"2026-08-06T19:23:15.946551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.090727Z","title":"Photorealistic text-to-image diffusion models with deep language understanding","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.090727Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:2442661aa61da01103e19eeba40d8bbd836acbdf904d19601700016410f5314a","observation_id":"ff1c0245-ef91-4394-9584-336d8bd4d49f","resolution":{"observed_at":"2026-08-06T19:23:15.090727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.917430Z","title":"Self- attention with relative position representations","venue":null,"work_id":"0dba3fa9-1e47-42f6-aaf8-12429be0a97a","year":2018},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.094653Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:367dc9d40a8841bcf99fb78848c3f54d8843353f2e7db069919cacb6d2b97f21","observation_id":"896e1bdb-a13f-4142-87fc-23a9935d0ea6","resolution":{"observed_at":"2026-08-06T19:23:15.921672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.098711Z","title":"Graph trans- formers: A survey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.098711Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:8c023acedca419e80cb94345573de02f0f0c8542ee25552276d956e71e2475d1","observation_id":"a164fb9f-4311-48e9-8676-66e56d22f493","resolution":{"observed_at":"2026-08-06T19:23:15.098711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.902434Z","title":"An empirical analysis on spatial reason- ing capabilities of large multimodal models","venue":null,"work_id":"d36233a8-838f-4948-80d6-e06f16828518","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.103044Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:aa2a8637695e429c88ce764d8e423643bcd8c7b25f3a397dc7026acfcb68eb74","observation_id":"f0bc5543-965e-46f5-b57c-361a5a8dff09","resolution":{"observed_at":"2026-08-06T19:23:15.907113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.887640Z","title":"Continual dif- fusion: Continual customization of text-to-image diffusion with c-lora","venue":null,"work_id":"2c385efc-9d95-43ca-a822-c0193ba72034","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.107488Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:d93b4d8a3ead27f0a4741ae0caf4ccaf7fa8c3f6e6d9549a9a09dcae432b3296","observation_id":"f309fc31-829f-438e-9785-bfed2adc4990","resolution":{"observed_at":"2026-08-06T19:23:15.892282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.873428Z","title":"Denois- ing diffusion implicit models","venue":null,"work_id":"6ceab89e-5cc4-4e39-b47d-8681e6c1493f","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.111442Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:3a91e2264d078fcb00e5604cc0f02e5733cc62d8b0c92b28431881f10401c6fa","observation_id":"232d6aea-8873-41e9-aa17-615ad04896b7","resolution":{"observed_at":"2026-08-06T19:23:15.877820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.859111Z","title":"Transformer-based image generation from scene graphs","venue":null,"work_id":"a34780fe-b848-4606-a86b-c095a84129e0","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.115379Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:cc5797c026e2261ad4171be6726a874a06019f44e6f7bc41f2850a740169875c","observation_id":"c8929f1b-89a5-4b9a-86aa-4305bfdb4187","resolution":{"observed_at":"2026-08-06T19:23:15.863788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.843902Z","title":"Reclip: A strong zero-shot baseline for referring expression compre- hension","venue":null,"work_id":"c127dbce-4bcd-49f3-9723-216a4b5f9f3e","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.119376Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:2217a6e66b90b10b2444eccdd50e63abfe109adf0ac102cf7f92c789cf75a9e8","observation_id":"3bfc6777-82d2-4515-943e-4e90e990b2d6","resolution":{"observed_at":"2026-08-06T19:23:15.848480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.828773Z","title":"Learning to compose dynamic tree structures for visual contexts","venue":null,"work_id":"5209093e-7247-46de-85a3-f8098b1bd726","year":2019},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.123341Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:fdde22994751cfd9df211bfbfb763b131d11128271e82a42f92f174408e31bea","observation_id":"5f93c6bd-4c6c-46dc-823f-ce3185f399f0","resolution":{"observed_at":"2026-08-06T19:23:15.832992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.815424Z","title":"Structured sparse r-cnn for di- rect scene graph generation","venue":null,"work_id":"b8bd7efe-7c20-420b-b0ef-92f8e332a60c","year":null},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.127626Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:6d098bd764bee217766790ccc71b1e88603b260752cfdf8f4e90b0a6160fc1ae","observation_id":"239300a6-bd98-44e7-b7ad-464e469cb0f3","resolution":{"observed_at":"2026-08-06T19:23:15.819616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.801819Z","title":"Belm: Bidirec- tional explicit linear multi-step sampler for exact inversion in diffusion models","venue":null,"work_id":"d09f7668-21c2-4a3d-ae6d-424dd495b037","year":2025},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.131788Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:06326d36d1d987eb57db3bb3beb573085481f52cc5225244c05fdb7bb78900a2","observation_id":"b16ec409-dbd3-43e7-807f-94325e226eb2","resolution":{"observed_at":"2026-08-06T19:23:15.805997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.788713Z","title":"Pair then relation: Pair-net for panoptic scene graph generation","venue":null,"work_id":"4450a9cb-8d01-4673-afe1-904cb2e1d7ba","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.136220Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:b1a1af86313d1744a4166890ce00db4e53c9880d4e6002f55249ddde4b276ccb","observation_id":"8cac48ea-bfeb-4125-9511-1ac332747907","resolution":{"observed_at":"2026-08-06T19:23:15.792970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.671386Z","title":"Stylediffusion: Controllable disentangled style transfer via diffusion models","venue":null,"work_id":"664dfb7c-088d-44fb-8d58-6a1661bea0c0","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.140503Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:7aebccad1aec118c70fad9067618708b2fe783edf04def98ce0811609fd25f46","observation_id":"7559965b-bac0-4096-96c5-ede4733a6a10","resolution":{"observed_at":"2026-08-06T19:23:15.676472Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.657593Z","title":"Chief: Clustering with higher-order motifs in big networks","venue":null,"work_id":"75594cb5-4b0d-4e5f-b265-fbab91523d9b","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.144674Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:57c433069a4924cdb4d1d1a4c8c07d0bbdbd044bb46f681c1887402752aa44e6","observation_id":"daa59dd5-6b14-4a9e-a843-504e35d0e036","resolution":{"observed_at":"2026-08-06T19:23:15.661663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.644499Z","title":"Scene graph generation by iterative message passing","venue":null,"work_id":"3d67df6f-f34d-4ef5-b459-313975ee2977","year":2017},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.148524Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:698b80efb56e2f0474e2d631f2e55fbe93cadedcad2d6e33ec91f6cb4b765d9d","observation_id":"1421948d-c5ef-4e45-a35a-bbe581f6e78c","resolution":{"observed_at":"2026-08-06T19:23:15.648602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.630886Z","title":"Scene graph generation by iterative message passing","venue":null,"work_id":"38ee48aa-3e74-498c-ae5c-d969caadd16e","year":2017},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.152769Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:9469939f099e12b1262c031e94270f55c3f17b9f80d1360590e38dd534521208","observation_id":"84bbab9b-e4b5-4ad6-83d1-aeebe41eb33e","resolution":{"observed_at":"2026-08-06T19:23:15.634811Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.617930Z","title":"Open-vocabulary panop- tic segmentation with text-to-image diffusion models","venue":null,"work_id":"5a84d3b3-86ab-43c7-9909-5b58ffa06e32","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.157189Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:0fea25a1dda84893f117ae08de42ce265f1aac2e230ab02447665e32fbcef6cf","observation_id":"927e7fc5-ca20-4a45-9c83-0bf07213b29c","resolution":{"observed_at":"2026-08-06T19:23:15.622102Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.603625Z","title":"Panoptic scene graph gen- eration","venue":null,"work_id":"fecf6476-261b-466b-8f92-2e8cac829258","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.161215Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:494aa10a5338dd1afe307b8522ecb31236fa10d3de874ce0084d9da602c3392f","observation_id":"1fc9baeb-56e2-4b10-a947-0abcfd5dacb1","resolution":{"observed_at":"2026-08-06T19:23:15.608251Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.590492Z","title":"Open-world human-object interaction detection via multi-modal prompts","venue":null,"work_id":"953141d2-71a1-44ef-9ad5-0027397addb6","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.165846Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:1519d15982d6d0c11cb6493eba49428f05d0ad3b27bd3f6a2f674c2691c781eb","observation_id":"902a1e79-b8af-4af6-a343-ea9534ce3109","resolution":{"observed_at":"2026-08-06T19:23:15.594901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.577209Z","title":"Visually-prompted language model for fine-grained scene graph generation in an open world","venue":null,"work_id":"68232850-b5c8-4e1d-b340-d600ad394d08","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.169859Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:9f50eba1ec17427037db0c76db4e64549d3730cc3b5eea35dbca3d692c063f0c","observation_id":"8e4df6da-59e2-4c70-822f-4c0ff33ff213","resolution":{"observed_at":"2026-08-06T19:23:15.581445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.563163Z","title":"Zero-shot scene graph generation with knowledge graph completion","venue":null,"work_id":"0f1406f1-5968-4438-8e18-db728ffc9468","year":2022},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.174133Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:d13c8951a4cbb2dc89db3b6a8198f450556ef3848e192af6d038c923cf238ee1","observation_id":"5d939c45-932a-4792-8759-bff0009cda44","resolution":{"observed_at":"2026-08-06T19:23:15.567311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.548719Z","title":"Graph transformer networks.Advances in neural information processing systems, 32, 2019","venue":null,"work_id":"7f7f3b0f-98c4-4211-b833-31f925168cd6","year":2019},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.178265Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:93e44d7f12fac2f60e1512eb2f2006ee1fccc77e82e20cc2cb4e40102061984e","observation_id":"8e7f760f-991e-459f-a5ae-6cb1bcb1b281","resolution":{"observed_at":"2026-08-06T19:23:15.553469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.532192Z","title":"Open-vocabulary object detection using captions","venue":null,"work_id":"5ca2cea2-7b3f-4cd6-9d15-3e7a618a80ce","year":2021},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.182074Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:3ce1570066b3cc120ed1bafb45f440b61370399c8ec880e72fb2633676b23b0f","observation_id":"b8c6cdae-8449-497f-8302-9afb75958538","resolution":{"observed_at":"2026-08-06T19:23:15.538385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.517301Z","title":"Neural motifs: Scene graph parsing with global con- text","venue":null,"work_id":"b75aa55f-074b-4398-8a3b-602ee640764d","year":2018},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.185933Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:c5865e84cdb14bd921ce13bbc4f79193931409408fad370b05a1932d834b5bd2","observation_id":"7bd12bce-0612-4ee6-b087-e9c50ff3e267","resolution":{"observed_at":"2026-08-06T19:23:15.521912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.501400Z","title":"gddim: Generalized denoising diffusion implicit models","venue":null,"work_id":"37a53416-b816-4527-9120-352dbbda12d4","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.189859Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:38b74f48073164ce353b4c5a07ee0182e35bbb5e154298bdf480e197de16276e","observation_id":"f2a338ee-7e70-421c-a661-584e46f95458","resolution":{"observed_at":"2026-08-06T19:23:15.505750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.484650Z","title":"Learning to generate language- supervised and open-vocabulary scene graph using pre- trained visual-semantic space","venue":null,"work_id":"ae0edc6f-4085-4068-b087-ea46bf1f7bed","year":null},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.193911Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:fb1cbc9ed66a3acc95034b4a61d51b1361116fc045fc43de0f113de462bb2edf","observation_id":"be0ccd3e-d26a-45eb-84df-4c2e1e4a53dd","resolution":{"observed_at":"2026-08-06T19:23:15.489519Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.470181Z","title":"Unleashing text-to-image diffusion models for visual perception","venue":null,"work_id":"5d921de1-be2d-4711-a406-278569707437","year":null},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.198533Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:d481072a03e4ec62d5ab6ea8592e931f0fad3634559d5a2bba3ea9b2ba3298d9","observation_id":"5b940a59-c365-4b65-bdc2-ffacc8599fe2","resolution":{"observed_at":"2026-08-06T19:23:15.474522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.456081Z","title":"Prototype-based embedding network for scene graph generation","venue":null,"work_id":"5b979233-ffcd-4438-954e-3fb3af20d335","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.203065Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:586407f1dbaeb660941481223f501e095b83b08ef4f9eb4bc9bf28cbf657ba97","observation_id":"51a6b8dd-58f2-443a-a97b-7e8724537c58","resolution":{"observed_at":"2026-08-06T19:23:15.460294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.441736Z","title":"Hilo: Ex- ploiting high low frequency relations for unbiased panoptic scene graph generation","venue":null,"work_id":"a614e736-b5e3-4cef-a8ad-eef341e62e9c","year":2023},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.207023Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:34516084a794eab7647e5ee59d27a33d8712b355b5a339cedd635885a779e88a","observation_id":"5879cd9c-68e7-451a-b7cb-4855e7770920","resolution":{"observed_at":"2026-08-06T19:23:15.446010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:23:15.425312Z","title":"Openpsg: Open-set panoptic scene graph generation via large multimodal models","venue":null,"work_id":"acd07e0c-3f75-4e13-aea3-73cc36a5595f","year":2024},"citing_paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T19:23:15.210670Z"},"links":{"citing_paper":"/paper/2507.05798"},"observation_digest":"sha256:3b3b911b2f443e673836c13a58567d10305fc4c6f366dbab92904038faccf0c0","observation_id":"c075dfd0-34e4-4bf1-99c0-0b6cf64573f2","resolution":{"observed_at":"2026-08-06T19:23:15.431443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.05798","last_updated":"2025-07-08T09:03:24Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-10T13:29:22.479120Z","submitted_at":"2025-07-08T09:03:24Z","title":"SPADE: Spatial-Aware Denoising Network for Open-vocabulary Panoptic Scene Graph Generation with Long- and Local-range Context Reasoning"},"reference_resolution":{"displayed":68,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":0,"verified_fuzzy":61},"total_outbound_references":68},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 68 of 68 outbound references and 0 inbound Pith citation observations for arXiv:2507.05798."}