{"as_of":"2026-08-13T10:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:69cfc7cb8e8d334f01786e8de4d7dbd7b05bbf21df5f9a3cfd6e2b59158f0e8b","coverage":[{"denominator":13,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":13,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T13:22:33.725821Z","state":"measured"},{"denominator":13,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":13,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.16310/citation-record","integrity":"/paper/2411.16310/integrity","json":"/paper/2411.16310/citation-record.json","paper":"/paper/2411.16310"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.851313Z","title":"Yolo-world: Real-time open- vocabulary object detection","venue":null,"work_id":"ff194b90-9d44-4959-99c5-8c4b85d0fdcb","year":2024},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.687440Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:5780c8e9e0bcb41515de008599d47a22e79d94ccbe3d90ad75709d699cdf1e69","observation_id":"c6439413-a6ed-41d9-a544-62203911048e","resolution":{"observed_at":"2026-08-12T13:22:33.854444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.842886Z","title":"Scannet: Richly- annotated 3d reconstructions of indoor scenes","venue":null,"work_id":"bb554a26-ca20-482f-bdee-493e50afdf1f","year":null},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.691006Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:7a23023d5cbff24fc8c4d4260ddf720e11847177c1cd6a114ecf390246a11d1f","observation_id":"eeb01ea7-238c-4aeb-a6ea-287ad6d2879a","resolution":{"observed_at":"2026-08-12T13:22:33.846061Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17146","last_updated":"2024-12-05T14:28:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-25T17:59:51Z","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.17146","snapshot_observed_at":"2026-08-12T13:22:33.694447Z","title":"Molmo and pixmo: Open weights and open data for state-of-the-art mul- timodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.694447Z"},"links":{"cited_paper":"/paper/2409.17146","citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:ba43ae87ae0d7ff3e020a5231774acb2974d40836ba35ca914047432e38755ed","observation_id":"c4d25a10-713b-4806-8b7b-b8cca2145354","resolution":{"observed_at":"2026-08-12T13:22:33.694447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.834247Z","title":"Scene- 5https : / / huggingface","venue":null,"work_id":"25c0a73d-d392-45fd-a87c-fcf4f1826057","year":2024},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.698108Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:ed584baf349a722758382a5a06a1ed4608bd1afa9aaa6472f0808436863b676e","observation_id":"3eb1dcb4-b1b4-4286-930d-fc1269af56a3","resolution":{"observed_at":"2026-08-12T13:22:33.837400Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.825866Z","title":"Openins3d: Snap and lookup for 3d open-vocabulary instance segmentation","venue":null,"work_id":"d53af1a3-5412-4fc1-b8f8-bf3dd9bbfdcd","year":2024},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.701161Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:856c6184501da3c5ead48924cd57f183d1d761b10d9bac05123f35a1529a1b00","observation_id":"a56f44a3-93b3-4254-be56-d9304eb65f94","resolution":{"observed_at":"2026-08-12T13:22:33.828944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.816998Z","title":"Lerf: Language embed- ded radiance fields","venue":null,"work_id":"16913e3c-bf00-4232-93a9-6dad2737157d","year":2023},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.704407Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:a009410da9e1e103e1226dafa5428b3aa943c82a415cd173efea16c59f211106","observation_id":"25d88995-0eaa-4d10-8b84-ce2efed795ec","resolution":{"observed_at":"2026-08-12T13:22:33.820149Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.808007Z","title":"Segment any- thing","venue":null,"work_id":"7b12d52e-431f-4fec-8e6d-f4d429606eae","year":2023},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.707530Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:e6881160c763d21b867a1537d2aa3765aa32ea406e2014cb4b8f828e48cd5c6b","observation_id":"8c7e7c2f-3dd6-4807-885e-2560dbf639e0","resolution":{"observed_at":"2026-08-12T13:22:33.810976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.799241Z","title":"Srinivasan, Matthew Tancik, Jonathan T","venue":null,"work_id":"e5150202-30dd-44d0-8ec5-f511ede29c5d","year":2021},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.710729Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:9647d36ce8081faa7d201512d90e71c7e7c945249fa3e144d533952a11fb877d","observation_id":"becfc2cd-3cbf-43a9-b6df-245cd713f30b","resolution":{"observed_at":"2026-08-12T13:22:33.802407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.790297Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"0edb52f2-348e-44fd-897d-00c61cca2f1a","year":2021},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.713698Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:912aded6f4c4b0d3686c24c25af4bad6c91123a09cfad66103af192574d2617d","observation_id":"d9444ab9-c448-40c4-b117-a4dfcac28967","resolution":{"observed_at":"2026-08-12T13:22:33.793678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.781614Z","title":"Mask3d: Mask trans- former for 3d semantic instance segmentation","venue":null,"work_id":"5d34f1ad-3144-4b6b-8b74-97b345b3444b","year":2023},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.716667Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:f07d3ddde93c81837cf07065389fcc0b37490de7ed2d73a84ec4c9b647106ffa","observation_id":"5a1d048b-19e2-4315-a587-588de74bb041","resolution":{"observed_at":"2026-08-12T13:22:33.784578Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.772987Z","title":"Openmask3d: Open-vocabulary 3d instance segmentation","venue":null,"work_id":"111a56bf-c997-4049-abe3-c3abcb7fa62f","year":2024},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.719773Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:0c1e2a43df1559aef72640a26440c49fcff9d4fc6b33a3654322cbd837b94b50","observation_id":"98d752c5-ac5e-4ded-a123-ac725516681a","resolution":{"observed_at":"2026-08-12T13:22:33.775967Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.722726Z","title":"Open-vocabulary panoptic segmentation with text-to-image diffusion models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.722726Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:4b54522772116b3db0d8cf110b8c14154d1fd3bd1c9477b7b966024ed3d03db6","observation_id":"b4014148-2ff8-4a70-aa79-387fc853ed6e","resolution":{"observed_at":"2026-08-12T13:22:33.722726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T13:22:33.757191Z","title":"Examples of selected contextual object masks with their scores, from highest rank (left) to lowest rank (right)","venue":null,"work_id":"2d3e63c7-a4bf-40df-bee4-b782ad77fb78","year":null},"citing_paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes","version":5},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-12T13:22:33.725821Z"},"links":{"citing_paper":"/paper/2411.16310"},"observation_digest":"sha256:4b68f03d3c212c2307c4fa9fc64657ca14410a9b26c2d504551ebc289bc85c7c","observation_id":"7920b1b8-361d-4793-be63-f16e0d4187a9","resolution":{"observed_at":"2026-08-12T13:22:33.761916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.16310","last_updated":"2025-05-28T07:37:44Z","latest_version":5,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T13:13:07.079103Z","submitted_at":"2024-11-25T11:57:48Z","title":"Functionality understanding and segmentation in 3D scenes"},"reference_resolution":{"displayed":13,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":2,"verified_exact":0,"verified_fuzzy":11},"total_outbound_references":13},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 13 of 13 outbound references and 0 inbound Pith citation observations for arXiv:2411.16310."}