{"as_of":"2026-08-07T15:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5dc3d1add64672af5b704d46d45c18df4049093d47e8e41477c7247032f23c64","coverage":[{"denominator":40,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:03:12.933913Z","state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.06687/citation-record","integrity":"/paper/2507.06687/integrity","json":"/paper/2507.06687/citation-record.json","paper":"/paper/2507.06687"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-3-642-03798-6","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:13.338360Z","title":"The Stixel World - A Compact Medium Level Representation of the 3D-World,","venue":null,"work_id":"b4e4ccc7-b4c5-41ab-bde5-e026c9942b6e","year":2009},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:09.234603Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:dfe15e83fd4254d1ff157cd11f03749ff3a791c8b171bca357c68d3022a8a8a1","observation_id":"90e10110-e03e-4790-8155-504f5079fcbe","resolution":{"observed_at":"2026-08-06T19:03:13.500083Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2407.08261","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.864903Z","title":"The AEIF Data Collection: A Dataset for Infrastructure-Supported Perception Research with Focus on Public Transportation,","venue":null,"work_id":"7eb1f760-b28b-48bf-be00-4d051f3425c6","year":2024},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:09.294332Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:d3b5266f85681966c8b049422e3ea17254256bac6d0509b3a3b2a3b783bed2c1","observation_id":"4dd7b530-40dc-4f78-9b02-12eb456dd12f","resolution":{"observed_at":"2026-08-06T19:03:14.869555Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"document/1058868","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.792443Z","title":"StixelNExT: Toward Monocular Low-Weight Perception for Object Segmentation and Free Space Detection,","venue":null,"work_id":"cf5238c1-d229-46db-a950-ee56dd7948aa","year":2024},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:09.366113Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:04539d1cb5156c6d46d0f657f66623445f11b43d49654813489c6fc8aeddfb66","observation_id":"516aa742-f6c4-4989-8725-26fb850c0290","resolution":{"observed_at":"2026-08-06T19:03:14.798239Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.910736Z","title":"Towards a Global Optimal Multi-Layer Stixel Representation of Dense 3D Data,","venue":null,"work_id":"2023be4a-14be-466f-819e-663fd9900b69","year":2011},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:09.413976Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:f94ab5674628da1a2778614a1cacdc12dbec75a064ccbf34230a2645691a41c4","observation_id":"5c0558c0-b669-47ae-831b-812ef9fa0688","resolution":{"observed_at":"2026-08-06T19:03:14.913300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"document/7535373","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.716088Z","title":"Semantic Stixels: Depth is not enough,","venue":null,"work_id":"1d60e171-6ee8-4cb5-ae25-1ea80df95ad7","year":2016},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:09.497755Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:5c31dd4c1ee9ee67b322f5a86690a82d7959370c8d1233c4ce5d092896b357d8","observation_id":"bd5d0b1c-28f1-4fcc-881c-cd6038ea3ff4","resolution":{"observed_at":"2026-08-06T19:03:14.719924Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"document/8814243","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.640923Z","title":"Instance Stixels: Segmenting and Grouping Stixels into Objects,","venue":null,"work_id":"1a4fc7c4-d480-43eb-9b85-286f3f650bfa","year":2019},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:09.569263Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:79d15b317036404caf8f3d000503548320aa457c123cbcd0d35a25e23bf9681f","observation_id":"59512be4-a5f5-424c-8ca0-8ae248298d3e","resolution":{"observed_at":"2026-08-06T19:03:14.646185Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.903792Z","title":"StixelNet: A Deep Convolutional Network for Obstacle Detection and Road Segmentation,","venue":null,"work_id":"84e9f7d3-8c8d-4634-9a39-2541c4bf1019","year":2015},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:09.648348Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:cca319009c71123eb18acb160132019e4371ce26d5d5a65a77dfe4d80ec5d8b3","observation_id":"f6866d5a-674b-4506-91ce-f1eea8ede548","resolution":{"observed_at":"2026-08-06T19:03:14.906345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"document/8265242","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.573088Z","title":"Real-Time Category- Based and General Obstacle Detection for Autonomous Driving,","venue":null,"work_id":"415284b4-bea2-4694-ae64-8cf2c00e1ce5","year":2017},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:09.741240Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:4f3a7b0e5b027c8306ca9ff4039d0e11e983af90835ce2c7e5b3c75204efa854","observation_id":"1f37f341-64fc-4d48-95b4-165c68eb1a43","resolution":{"observed_at":"2026-08-06T19:03:14.577232Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.02635","last_updated":"2019-08-07T13:54:51Z","snapshot_observed_at":"2026-08-07T10:17:05.638818Z","submitted_at":"2019-08-07T13:54:51Z","title":"Mono-Stixels: Monocular depth reconstruction of dynamic street scenes","version":1},"cited_work":{"arxiv_id":"1908.02635","doi":null,"metadata_source":"pith","pith_arxiv_id":"1908.02635","snapshot_observed_at":"2026-08-06T19:03:14.481771Z","title":"Mono-Stixels: Monocular depth reconstruction of dynamic street scenes","venue":"cs.CV","work_id":"9631f483-0b74-4079-a5a9-a4776979ca71","year":2019},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:09.813984Z"},"links":{"cited_paper":"/paper/1908.02635","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:6d83fad7b0576e63f8c456b95300daa819125626b73e0d378bf5481430e33cc5","observation_id":"37836f17-e459-46ff-9b92-383c3ca1ccdf","resolution":{"observed_at":"2026-08-06T19:03:14.485792Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-3-030-11009-3","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:13.065518Z","title":"Exploiting Single Image Depth Prediction for Mono-stixel Estimation,","venue":null,"work_id":"5b081a52-564f-4e77-b192-fbde3268946e","year":2018},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:09.869727Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:d0ee43d227e4ac53f0986584ac8a81c2d207d5d83b67e1bd9a0051b9419034f6","observation_id":"bdb2a323-4e1b-4a84-97f8-6228be36dfce","resolution":{"observed_at":"2026-08-06T19:03:13.181365Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.896266Z","title":"Unsupervised Monocular Depth Estimation with Left-Right Consistency,","venue":null,"work_id":"515c5dfb-f008-4b28-8826-cd5ffbeb8d89","year":2017},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:09.954295Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:713c562a70224f5ec620085f24d4670b242b520e0611aa578323d3b47b55f8b2","observation_id":"aba5385b-fc50-40d5-b176-1146ed704cbc","resolution":{"observed_at":"2026-08-06T19:03:14.899159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1806.01260","last_updated":"2019-08-17T22:57:30Z","snapshot_observed_at":"2026-08-04T09:21:24.573736Z","submitted_at":"2018-06-04T17:58:05Z","title":"Digging Into Self-Supervised Monocular Depth Estimation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1806.01260","snapshot_observed_at":"2026-08-06T19:03:10.129567Z","title":"Digging Into Self-Supervised Monocular Depth Estimation,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.129567Z"},"links":{"cited_paper":"/paper/1806.01260","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:f4c6e769046db0e68a352dc3645ea762835d11e0d8c210d1c4e444b06692463b","observation_id":"62ecf9bc-591f-479e-bb2d-0d60479bda0e","resolution":{"observed_at":"2026-08-06T19:03:10.129567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10891","last_updated":"2024-04-07T06:52:21Z","snapshot_observed_at":"2026-08-01T22:50:12.319180Z","submitted_at":"2024-01-19T18:59:52Z","title":"Depth Anything: Unleashing the Power of Large-Scale Unlabeled Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.10891","snapshot_observed_at":"2026-08-06T19:03:10.196940Z","title":"Depth Anything: Unleashing the Power of Large-Scale Unlabeled Data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.196940Z"},"links":{"cited_paper":"/paper/2401.10891","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:02dad7c9a61d8dea4ff4737c4a6fa94e2aacc305aca79dddfd090f15e498ae55","observation_id":"7f796b17-4330-4239-b65f-20fa3bcc3bca","resolution":{"observed_at":"2026-08-06T19:03:10.196940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.889452Z","title":"Depth Anything V2,","venue":null,"work_id":"d78f86d9-fd60-487e-8dbd-dc0ca2a7598c","year":null},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.277548Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:bdb39cf5b098e7cfdb1806dc33d965dc4e0206959857009b6020aee82c535a4d","observation_id":"e40cdcb8-d5ff-4d00-bef8-28979cbdc23f","resolution":{"observed_at":"2026-08-06T19:03:14.891752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.00726","last_updated":"2022-03-29T17:59:46Z","snapshot_observed_at":"2026-08-01T22:00:39.020096Z","submitted_at":"2021-12-01T18:59:57Z","title":"MonoScene: Monocular 3D Semantic Scene Completion","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.00726","snapshot_observed_at":"2026-08-06T19:03:10.417954Z","title":"MonoScene: Monocular 3D Semantic Scene Completion,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.417954Z"},"links":{"cited_paper":"/paper/2112.00726","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:e9dfe5710d15e005d703a2ec1d49558e29581dfe8c1bfe0e1c521bcf6b3a8e9d","observation_id":"fb95f63c-1092-49d3-97e9-3c3b32c14bc7","resolution":{"observed_at":"2026-08-06T19:03:10.417954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.15694","last_updated":"2023-05-25T04:03:46Z","snapshot_observed_at":"2026-07-31T03:00:36.813042Z","submitted_at":"2023-05-25T04:03:46Z","title":"Learning Occupancy for Monocular 3D Object Detection","version":1},"cited_work":{"arxiv_id":"2305.15694","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.15694","snapshot_observed_at":"2026-08-06T19:03:14.432451Z","title":"Learning Occupancy for Monocular 3D Object Detection","venue":"cs.CV","work_id":"8ce54e6b-dd60-42fe-9f63-9a54ea1044c5","year":2023},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.487100Z"},"links":{"cited_paper":"/paper/2305.15694","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:014aca4d5a7ea85bb61cd6e4d9aa164e1b7a0f31e941b4f402d70d4417a2092d","observation_id":"628bc0ed-1ad3-4d66-9241-f830d95fdb78","resolution":{"observed_at":"2026-08-06T19:03:14.435055Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1811.10247","last_updated":"2020-03-31T14:52:26Z","snapshot_observed_at":"2026-07-06T07:16:59.694382Z","submitted_at":"2018-11-26T09:36:40Z","title":"MonoGRNet: A Geometric Reasoning Network for Monocular 3D Object Localization","version":2},"cited_work":{"arxiv_id":"1811.10247","doi":null,"metadata_source":"pith","pith_arxiv_id":"1811.10247","snapshot_observed_at":"2026-08-06T19:03:14.422346Z","title":"MonoGRNet: A Geometric Reasoning Network for Monocular 3D Object Localization","venue":"cs.CV","work_id":"12e9ae7a-ec81-49e1-a9a2-4fad17fa04ed","year":2018},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.575801Z"},"links":{"cited_paper":"/paper/1811.10247","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:fdc5bc1052fe2ea8694bdc3f376a280a540c61229f605d301a95b84358d1ed7c","observation_id":"b12cd330-b73a-4490-869c-10acf423ae34","resolution":{"observed_at":"2026-08-06T19:03:14.425214Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.15319","last_updated":"2024-01-27T06:45:35Z","snapshot_observed_at":"2026-07-06T17:21:14.265076Z","submitted_at":"2024-01-27T06:45:35Z","title":"You Only Look Bottom-Up for Monocular 3D Object Detection","version":1},"cited_work":{"arxiv_id":"2401.15319","doi":null,"metadata_source":"pith","pith_arxiv_id":"2401.15319","snapshot_observed_at":"2026-08-06T19:03:14.411163Z","title":"You Only Look Bottom-Up for Monocular 3D Object Detection","venue":"cs.CV","work_id":"e9c2cfac-701a-4283-ada3-cd012d2a316f","year":2024},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.652984Z"},"links":{"cited_paper":"/paper/2401.15319","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:f6de7200cc0c04030e92645f3691f9c0d7221240d76eecb6192ac8fd6cebf3fa","observation_id":"9fd72b2d-7fe1-4793-9cae-789375e232a7","resolution":{"observed_at":"2026-08-06T19:03:14.414239Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.00966","last_updated":"2022-03-30T17:27:23Z","snapshot_observed_at":"2026-08-03T08:43:51.471413Z","submitted_at":"2021-10-03T09:52:46Z","title":"Translating Images into Maps","version":2},"cited_work":{"arxiv_id":"2110.00966","doi":null,"metadata_source":"pith","pith_arxiv_id":"2110.00966","snapshot_observed_at":"2026-08-06T19:03:14.401098Z","title":"Translating Images into Maps","venue":"cs.CV","work_id":"6e518b6e-d576-41f8-a1b6-764918d0f955","year":2021},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.833228Z"},"links":{"cited_paper":"/paper/2110.00966","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:51e88a5ebb699a6e33c77d559e678a27158fa2354982a474f6e978354e541eb8","observation_id":"e2eda1a9-b5b0-40e5-bc91-742ca2752f9a","resolution":{"observed_at":"2026-08-06T19:03:14.403757Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.881912Z","title":"SeaBird: Segmentation in Bird’s View with Dice Loss Improves Monocular 3D Detection of Large Objects,","venue":null,"work_id":"f5e037c9-7d53-4a90-855b-96d46f2ef49c","year":2024},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.903057Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:b0f10f07f6237d793dc1329f3068f48b994c5eadc248ed4ad75387ecbf160ad1","observation_id":"625fd17c-3ade-4ea9-95b8-16811e295a75","resolution":{"observed_at":"2026-08-06T19:03:14.884936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.06093","last_updated":"2024-03-10T04:38:27Z","snapshot_observed_at":"2026-07-06T17:42:08.056523Z","submitted_at":"2024-03-10T04:38:27Z","title":"Enhancing 3D Object Detection with 2D Detection-Guided Query Anchors","version":1},"cited_work":{"arxiv_id":"2403.06093","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.06093","snapshot_observed_at":"2026-08-06T19:03:14.390179Z","title":"Enhancing 3D Object Detection with 2D Detection-Guided Query Anchors","venue":"cs.CV","work_id":"0c85c246-02e8-4314-8152-f718ca99bef7","year":2024},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.988247Z"},"links":{"cited_paper":"/paper/2403.06093","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:c87fecb2fbcfa9f6ac5b3764bb0bd64ceab463572c50be6bd38ca1e775499839","observation_id":"fb920c05-bf22-4bf7-9972-fbbf8d52bb91","resolution":{"observed_at":"2026-08-06T19:03:14.393526Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"document/8010878","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.377474Z","title":"Estimating Depth From Monocular Images as Classification Using Deep Fully Convolutional Residual Networks,","venue":null,"work_id":"dd2b45ed-5f40-415f-86c1-2a5cab6d549f","year":2018},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:11.064081Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:99a69c7ac544761f8863dff4d30e8bfe033e2157792c4f790724013940614a9b","observation_id":"e1235f63-0eaf-473e-963a-85b459d7e130","resolution":{"observed_at":"2026-08-06T19:03:14.381969Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"document/8954293","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.318218Z","title":"Pseudo-LiDAR From Visual Depth Estimation: Bridging the Gap in 3D Object Detection for Autonomous Driving,","venue":null,"work_id":"fbdfb48d-cfb5-4377-b827-f1eab1e2dff0","year":2019},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:11.138354Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:0fc153055d2eab7882ea6a4c384fa598a0376a223ab47c1170c19dd997bbbdca","observation_id":"4f19a439-b83f-461c-81e0-e9801a544e44","resolution":{"observed_at":"2026-08-06T19:03:14.322618Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.02028","last_updated":"2019-04-03T14:31:35Z","snapshot_observed_at":"2026-07-06T07:43:38.480475Z","submitted_at":"2019-04-03T14:31:35Z","title":"CAM-Convs: Camera-Aware Multi-Scale Convolutions for Single-View Depth","version":1},"cited_work":{"arxiv_id":"1904.02028","doi":null,"metadata_source":"pith","pith_arxiv_id":"1904.02028","snapshot_observed_at":"2026-08-06T19:03:14.259656Z","title":"CAM-Convs: Camera-Aware Multi-Scale Convolutions for Single-View Depth","venue":"cs.CV","work_id":"90516e52-0643-46c9-be30-77c93b20b88b","year":2019},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:11.205012Z"},"links":{"cited_paper":"/paper/1904.02028","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:1702132959b19d7ccb7cde2bc9d10be98863d43f73ef53c02602f51fa66740b2","observation_id":"bf99724e-e5de-4fb3-bf85-afffa3947b84","resolution":{"observed_at":"2026-08-06T19:03:14.262352Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.10039","last_updated":"2018-03-27T12:26:15Z","snapshot_observed_at":"2026-07-06T06:30:26.313049Z","submitted_at":"2018-03-27T12:26:15Z","title":"Learning Depth from Single Images with Deep Neural Network Embedding Focal Length","version":1},"cited_work":{"arxiv_id":"1803.10039","doi":null,"metadata_source":"pith","pith_arxiv_id":"1803.10039","snapshot_observed_at":"2026-08-06T19:03:14.177757Z","title":"Learning Depth from Single Images with Deep Neural Network Embedding Focal Length","venue":"cs.CV","work_id":"79a576ce-beb2-4ef3-add6-3e496f1c9603","year":2018},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:11.309790Z"},"links":{"cited_paper":"/paper/1803.10039","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:af05dc346a7b9b2747357941a209d1da902ab3d8647aaae9c4761a710e516651","observation_id":"e20e83fc-db50-4822-a1cd-7b482e35491e","resolution":{"observed_at":"2026-08-06T19:03:14.251338Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"document/9981561","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:13.984092Z","title":"Patchwork++: Fast and Robust Ground Segmentation Solving Partial Under-Segmentation Using 3D Point Cloud,","venue":null,"work_id":"7ca21372-9cdd-45f3-8e3a-89df635e1d9a","year":2022},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:11.406433Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:d287ebaa1dc1dfb433a5192125edabfd253a11a7429a0356f46132a764a499c5","observation_id":"83e9145d-e30c-4845-853b-2d7e3c7b9805","resolution":{"observed_at":"2026-08-06T19:03:14.048991Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.03545","last_updated":"2022-03-02T15:08:16Z","snapshot_observed_at":"2026-07-06T12:26:09.094759Z","submitted_at":"2022-01-10T18:59:10Z","title":"A ConvNet for the 2020s","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.03545","snapshot_observed_at":"2026-08-06T19:03:11.481155Z","title":"A ConvNet for the 2020s,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:11.481155Z"},"links":{"cited_paper":"/paper/2201.03545","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:9906912b463d47096cae89a56ba1d684c48d0d7f85135990c993dfb8a8afc952","observation_id":"9d771b6d-f481-40e7-84cf-b33cb4af27b5","resolution":{"observed_at":"2026-08-06T19:03:11.481155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:14.874436Z","title":"Hartley and A","venue":null,"work_id":"6a328eb9-e8d9-4786-8030-586877df216c","year":2004},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:11.563217Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:4872d92dbda4b76429d1920492361b07577427023e7e0a76119fe47e5890ccd8","observation_id":"d4855c8e-2ce2-4ee6-a4ac-d20100adddd8","resolution":{"observed_at":"2026-08-06T19:03:14.876685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1912.04838","last_updated":"2020-05-12T23:28:05Z","snapshot_observed_at":"2026-07-06T08:43:28.341292Z","submitted_at":"2019-12-10T17:28:55Z","title":"Scalability in Perception for Autonomous Driving: Waymo Open Dataset","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.04838","snapshot_observed_at":"2026-08-06T19:03:11.593133Z","title":"Scalability in Perception for Autonomous Driving: Waymo Open Dataset,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:11.593133Z"},"links":{"cited_paper":"/paper/1912.04838","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:a1b325699b1f32964c6de0d6a259bdb3909e372a9274cc29375f3f783c3f87fd","observation_id":"89fdc2d9-2465-489c-aa35-0db7f0d22418","resolution":{"observed_at":"2026-08-06T19:03:11.593133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.07705","last_updated":"2024-05-03T19:00:47Z","snapshot_observed_at":"2026-07-06T13:21:17.244944Z","submitted_at":"2022-06-15T17:57:41Z","title":"LET-3D-AP: Longitudinal Error Tolerant 3D Average Precision for Camera-Only 3D Detection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.07705","snapshot_observed_at":"2026-08-06T19:03:11.663731Z","title":"LET-3D-AP: Longitudinal Error Tolerant 3D Average Precision for Camera- Only 3D Detection,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:11.663731Z"},"links":{"cited_paper":"/paper/2206.07705","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:8bdaa1c8d689a6aaea1039e950a78228746f6e424d3b9f16ca3b433ab01e5c0a","observation_id":"57e63167-52b8-4157-8446-9a4c7e99de75","resolution":{"observed_at":"2026-08-06T19:03:11.663731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:11.854501Z","title":"Are we ready for autonomous driving? The KITTI vision benchmark suite,","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:11.854501Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:1572801b5b8357d1b2ca8e26c1d256a5ab485899d4eac9bbfc9617104453980a","observation_id":"419b70d6-a265-4877-b208-bab3e1f686f3","resolution":{"observed_at":"2026-08-06T19:03:11.854501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.14160","last_updated":"2021-11-25T08:20:46Z","snapshot_observed_at":"2026-07-06T11:33:49.489122Z","submitted_at":"2021-07-29T16:30:33Z","title":"Probabilistic and Geometric Depth: Detecting Objects in Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.14160","snapshot_observed_at":"2026-08-06T19:03:12.007827Z","title":"Probabilistic and Geometric Depth: Detecting Objects in Perspective,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:12.007827Z"},"links":{"cited_paper":"/paper/2107.14160","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:bdf8f1bd8df87ecc5b7017e6285606d161ebbad0c543531007687bf0b912a0e9","observation_id":"597088ce-675c-4c40-90c4-5894ca8aa5c9","resolution":{"observed_at":"2026-08-06T19:03:12.007827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:03:12.207041Z","title":"The Pascal Visual Object Classes Challenge: A Retrospective,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:12.207041Z"},"links":{"citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:19b4b0488633d21087888e54d9d22c2313238c74ab8b855b99d04bc4d00727fd","observation_id":"1315b551-ee4a-4ef4-8bb7-84d5b75cf0a9","resolution":{"observed_at":"2026-08-06T19:03:12.207041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.00298","last_updated":"2021-06-23T22:04:56Z","snapshot_observed_at":"2026-08-07T10:03:50.567074Z","submitted_at":"2021-04-01T07:08:36Z","title":"EfficientNetV2: Smaller Models and Faster Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.00298","snapshot_observed_at":"2026-08-06T19:03:12.361797Z","title":"EfficientNetV2: Smaller Models and Faster Training,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:12.361797Z"},"links":{"cited_paper":"/paper/2104.00298","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:4d5dfa1e5470ef9ac1eb1611154056abaa58851c820ff7c1a96344f58b32614e","observation_id":"16340daa-e2be-4f96-b20e-cf644f51a53d","resolution":{"observed_at":"2026-08-06T19:03:12.361797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.02244","last_updated":"2019-11-20T17:26:40Z","snapshot_observed_at":"2026-07-06T07:50:45.045024Z","submitted_at":"2019-05-06T19:38:31Z","title":"Searching for MobileNetV3","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.02244","snapshot_observed_at":"2026-08-06T19:03:12.515597Z","title":"Searching for MobileNetV3,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:12.515597Z"},"links":{"cited_paper":"/paper/1905.02244","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:93771e6104ac98efc4ebf0e2056852adfa069dcdeeba16ca57653802fd9f335f","observation_id":"220cab35-4ea2-4dbe-8b98-5ad572a68761","resolution":{"observed_at":"2026-08-06T19:03:12.515597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1807.11164","last_updated":"2018-07-30T04:18:25Z","snapshot_observed_at":"2026-07-06T06:52:58.479955Z","submitted_at":"2018-07-30T04:18:25Z","title":"ShuffleNet V2: Practical Guidelines for Efficient CNN Architecture Design","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1807.11164","snapshot_observed_at":"2026-08-06T19:03:12.663510Z","title":"ShuffleNet V2: Practical Guidelines for Efficient CNN Architecture Design,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:12.663510Z"},"links":{"cited_paper":"/paper/1807.11164","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:b8fcf54e2adce6f028df064a51361c6e09b203d4d52706f097ebb498a97e258a","observation_id":"5f12e28f-e63b-4a86-aad5-e32549a41d18","resolution":{"observed_at":"2026-08-06T19:03:12.663510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.14030","last_updated":"2021-08-17T16:41:34Z","snapshot_observed_at":"2026-08-07T00:32:15.123562Z","submitted_at":"2021-03-25T17:59:31Z","title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.14030","snapshot_observed_at":"2026-08-06T19:03:12.809376Z","title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:12.809376Z"},"links":{"cited_paper":"/paper/2103.14030","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:5d394df1e955b06eff382cfdc5cc0336881b2cc504a26591d5d9849a1d9f7d75","observation_id":"f4e376cc-2d94-443f-a733-0b8fff9d5d28","resolution":{"observed_at":"2026-08-06T19:03:12.809376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1706.03762","last_updated":"2023-08-02T00:41:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-06-12T17:57:34Z","title":"Attention Is All You Need","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.03762","snapshot_observed_at":"2026-08-06T19:03:12.933913Z","title":"Attention Is All You Need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:12.933913Z"},"links":{"cited_paper":"/paper/1706.03762","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:55657e28e830af41ff650322d883aa48fedffa02fae0b89429060ed9375503a9","observation_id":"769467a0-083f-428c-b36e-0770f2855011","resolution":{"observed_at":"2026-08-06T19:03:12.933913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1609.03677","last_updated":"2017-04-12T14:40:50Z","snapshot_observed_at":"2026-07-06T05:10:34.491565Z","submitted_at":"2016-09-13T04:48:31Z","title":"Unsupervised Monocular Depth Estimation with Left-Right Consistency","version":3},"cited_work":{"arxiv_id":"1609.03677","doi":null,"metadata_source":"pith","pith_arxiv_id":"1609.03677","snapshot_observed_at":"2026-08-06T19:03:14.470261Z","title":"Unsupervised Monocular Depth Estimation with Left-Right Consistency","venue":"cs.CV","work_id":"82187b82-8b59-42fa-98cf-0061e372b750","year":2016},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.062771Z"},"links":{"cited_paper":"/paper/1609.03677","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:b2199a228d664c189f2f94ff0e18a04f39595df3c551158830c54503d0d84e56","observation_id":"d152aa18-e1e2-4c3f-96b8-f4e4acdbc324","resolution":{"observed_at":"2026-08-06T19:03:14.473357Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09414","last_updated":"2024-10-20T11:24:09Z","snapshot_observed_at":"2026-07-06T18:30:32.982860Z","submitted_at":"2024-06-13T17:59:56Z","title":"Depth Anything V2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09414","snapshot_observed_at":"2026-08-06T19:03:10.346021Z","title":"Available: http://arxiv.org/abs/2406.09414","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T19:03:10.346021Z"},"links":{"cited_paper":"/paper/2406.09414","citing_paper":"/paper/2507.06687"},"observation_digest":"sha256:30846d466cf5c1aff7995f19a482518834c05a68bb345c84f65384c3af4cff3a","observation_id":"07468896-3aad-4367-a6b7-06bef8b499a4","resolution":{"observed_at":"2026-08-06T19:03:10.346021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.06687","last_updated":"2025-07-09T09:30:07Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T00:32:46.739383Z","submitted_at":"2025-07-09T09:30:07Z","title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception"},"reference_resolution":{"displayed":40,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":15,"verified_exact":18,"verified_fuzzy":6},"total_outbound_references":40},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 40 of 40 outbound references and 0 inbound Pith citation observations for arXiv:2507.06687."}