{"as_of":"2026-08-14T20:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3d5617063c5509356790ffc73d2008bcee7023c724e644b7b4d57845c1bc1a53","coverage":[{"denominator":76,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":76,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T19:48:32.707991Z","state":"measured"},{"denominator":76,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":76,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.10334/citation-record","integrity":"/paper/2411.10334/integrity","json":"/paper/2411.10334/citation-record.json","paper":"/paper/2411.10334"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1803.01164","last_updated":"2018-09-12T15:57:23Z","snapshot_observed_at":"2026-08-14T19:40:00.112956Z","submitted_at":"2018-03-03T13:46:40Z","title":"The History Began from AlexNet: A Comprehensive Survey on Deep Learning Approaches","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.01164","snapshot_observed_at":"2026-08-12T19:48:31.546778Z","title":"The history began from alexnet: A comprehensive survey on deep learning approaches","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.546778Z"},"links":{"cited_paper":"/paper/1803.01164","citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:524c8592a39ea08047e04a3fc1c5cab0d869c94d132b0ad5cc3c6598e71f8946","observation_id":"bf177382-30de-4a6a-b86d-b36b39267cb5","resolution":{"observed_at":"2026-08-12T19:48:31.546778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.742804Z","title":null,"venue":null,"work_id":"40bd0075-2fd1-4952-9d8b-3ef7b7abb3ea","year":2018},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.563958Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:ff7b77fcb88aa21f90e31bec1e3a9a3ddbdb2985c0c9ce13a4ca7dd19427cd16","observation_id":"0da74ba3-62e7-46ba-9df8-ec345c8790da","resolution":{"observed_at":"2026-08-12T19:48:34.766853Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.12288","last_updated":"2023-02-23T19:13:10Z","snapshot_observed_at":"2026-07-06T14:55:15.719380Z","submitted_at":"2023-02-23T19:13:10Z","title":"ZoeDepth: Zero-shot Transfer by Combining Relative and Metric Depth","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.12288","snapshot_observed_at":"2026-08-12T19:48:31.567333Z","title":"Zoedepth: Zero-shot trans- fer by combining relative and metric depth","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.567333Z"},"links":{"cited_paper":"/paper/2302.12288","citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:768f7ea5fc84935f434cfe8e157b952f4b5981e3ddc961e78e3e2d3e5528a506","observation_id":"d5ecf148-3001-4698-b3d5-f835d95c5085","resolution":{"observed_at":"2026-08-12T19:48:31.567333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.646634Z","title":"Multimodal fu- sion via teacher-student network for indoor action recogni- tion","venue":null,"work_id":"9df7139f-134b-49cd-b5b0-951ffef7ebe4","year":2021},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.571113Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:a575a30464424730495acb6745dd6f5b774042b69113f0096b99b25f12e0a28a","observation_id":"a22820ec-8854-4e2f-b6d1-ff6ecbaabe75","resolution":{"observed_at":"2026-08-12T19:48:34.676976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.637852Z","title":"Realtime multi-person 2d pose estimation using part affinity fields","venue":null,"work_id":"06628fa6-3f2e-4cf5-912c-e2fe2d60061f","year":2017},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.574730Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:47d7c7456197ee07dfd44e9289180741195f0cab60830972751583124aa28b2f","observation_id":"a5a36342-85d7-4dff-8a7e-ed3991173067","resolution":{"observed_at":"2026-08-12T19:48:34.641110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:31.578084Z","title":"Emerg- ing properties in self-supervised vision transformers","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.578084Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:21804ae78b76a3d1de60d9fef7c1fbf551a938436f1e3c81750d0d936f48ee3a","observation_id":"7b84e4ff-1d62-4062-ad08-39fdd4a17794","resolution":{"observed_at":"2026-08-12T19:48:31.578084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:31.653336Z","title":"Higherhrnet: Scale- aware representation learning for bottom-up human pose es- timation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.653336Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:b9bfd44362de95d01b7e2440efca0a0a33255adcb0336d2d620c00697a0579b9","observation_id":"80cec3df-a0ad-418c-8c13-2ce3ad53992e","resolution":{"observed_at":"2026-08-12T19:48:31.653336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1406.1078","last_updated":"2014-09-03T00:25:02Z","snapshot_observed_at":"2026-07-06T03:45:28.546418Z","submitted_at":"2014-06-03T17:47:08Z","title":"Learning Phrase Representations using RNN Encoder-Decoder for Statistical Machine Translation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1406.1078","snapshot_observed_at":"2026-08-12T19:48:31.717798Z","title":"Learning phrase representations using rnn encoder-decoder for statistical machine translation","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.717798Z"},"links":{"cited_paper":"/paper/1406.1078","citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:0c0593212f708d130c7ed919af0780fe5b121b568e0ff64745dedecdc378f278","observation_id":"f1a1f4ff-922d-4415-9b13-3faab0d30d20","resolution":{"observed_at":"2026-08-12T19:48:31.717798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.619166Z","title":"Keras documentation - model checkpoint","venue":null,"work_id":"5dae84f4-d004-4a05-8688-3c25d71488bf","year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.778044Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:521b4470e7d855eeb239fa17762f6faa48430a3aa25a719e3a110eba22c1e73e","observation_id":"a9c51e9b-3c38-4b90-879e-d988bce00bdd","resolution":{"observed_at":"2026-08-12T19:48:34.622847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.610033Z","title":"Keras documentation - early stopping","venue":null,"work_id":"8d1763ee-0f65-4aea-bb82-3a735c85c8dc","year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.781155Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:78530a3be7e367b7e8cbff95222d3540a42eff0adfe92da5a3ab1bbfa3200e52","observation_id":"74b7c40f-cff7-47c0-a08a-efac74d5d06c","resolution":{"observed_at":"2026-08-12T19:48:34.613391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.600702Z","title":"Unihcp: A unified model for human-centric perceptions","venue":null,"work_id":"abb4a4d1-606a-44d9-9e58-190c42e354b6","year":2023},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.786558Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:2fafcb10a899a54ad6223be78f09833c99edc7cd807c8b9f7ca0527ff51b0b2b","observation_id":"dc5e7b47-77b0-4f3f-9d43-dea9600589e9","resolution":{"observed_at":"2026-08-12T19:48:34.604168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:31.791114Z","title":"Adam: A method for stochastic opti- mization","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.791114Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:0ac648020edca92371a8a1895f1852ae6b1476dc682804845d3bf8afcbe5489a","observation_id":"a48e97ad-a07a-401c-ae1d-8c995dccc199","resolution":{"observed_at":"2026-08-12T19:48:31.791114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T14:19:26.598265Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-12T19:48:31.795196Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.795196Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:5d4ac63541664a6693d45e3af685e30940b2bc73f1b2408a5f519e442258d5bd","observation_id":"3c3e908e-6545-46d1-9c17-0fabbd72a5c1","resolution":{"observed_at":"2026-08-12T19:48:31.795196Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.516110Z","title":null,"venue":null,"work_id":"502fe119-c6e8-4747-b620-e993a3f51f02","year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.800000Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:e3452e9a0e828e3466ff926708c1c6e6649eeee63837cc12dd845114c7379e8e","observation_id":"73770d07-986b-4b96-a4c5-1cc61453b9f4","resolution":{"observed_at":"2026-08-12T19:48:34.557720Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.455062Z","title":null,"venue":null,"work_id":"88514203-2f83-477f-9c62-b08e654835c1","year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.803133Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:0748f500ed9cb9e7df12266e040a103f6ecb7f52a9ca534b0b26095d411700b4","observation_id":"fb4c3631-e627-43b2-b332-dbd6c8cb03a2","resolution":{"observed_at":"2026-08-12T19:48:34.509548Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.363918Z","title":"A stop list for general text","venue":null,"work_id":"1f6f8e7e-94af-4820-89c5-83c12792fac8","year":1989},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.806265Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:2e7b018d49bdcf3ecec20342a6ab7f0bcced575fcb3b7a777f41450acb3bfbd6","observation_id":"ccd16e5d-00f6-4ca1-b584-6d7b6967377c","resolution":{"observed_at":"2026-08-12T19:48:34.367547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.334491Z","title":"Visiongpt2","venue":null,"work_id":"833c7d7a-fa07-4669-9bdc-90a85a607aec","year":2023},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.849735Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:cb8ae83e0f031919e02d9992f803c59aad154552fd28d98e0f9a0c642a603c7a","observation_id":"aecde1c8-a446-4357-9786-b221ac1ae827","resolution":{"observed_at":"2026-08-12T19:48:34.357075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:31.899241Z","title":"Understanding the diffi- culty of training deep feedforward neural networks","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.899241Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:9cb65ab95ce6d10fa739edf85c6b9af28c50b4ad58e9195d7a9cb92ef5ac86af","observation_id":"4f07817a-9c7c-4b12-b8dd-dc52422b2760","resolution":{"observed_at":"2026-08-12T19:48:31.899241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.178788Z","title":"Densepose: Dense human pose estimation in the wild","venue":null,"work_id":"2f1e9dcf-77d1-4a3e-97cc-bcbfa783c64d","year":2018},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.902904Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:c976feee0c9256b65970649ec63c6a9a01be7ac259e87c5e44756f15b08f99b7","observation_id":"12e2bc46-d8fc-4feb-8cd3-37f419ce8229","resolution":{"observed_at":"2026-08-12T19:48:34.267321Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:31.906285Z","title":"Deep residual learning for image recognition","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.906285Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:6a8ba409ac17f8ede81d424560f0755794187a0a0ee9b47e9e6d88cef7c8a8f4","observation_id":"638306de-22f7-430d-956d-c956bf1ab407","resolution":{"observed_at":"2026-08-12T19:48:31.906285Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:31.909520Z","title":"Mask r-cnn","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.909520Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:a7506e569f9a57338824969f670415c447983fadaf8425e9ad18061fec9100ef","observation_id":"3111ff37-4590-4d2b-a203-69a9a6282961","resolution":{"observed_at":"2026-08-12T19:48:31.909520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.159827Z","title":"Long short-term memory","venue":null,"work_id":"c28840cc-75d9-41e3-a856-6f77315bdf33","year":1997},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.912656Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:2ef349bbb9a372415170ecf86738d868ab34af8050ee71a8cb6e522dab881e1a","observation_id":"3a88dbba-1b20-4c14-96a2-a4f869f95bdc","resolution":{"observed_at":"2026-08-12T19:48:34.163039Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.025335Z","title":"A multi-instance multi-label dual learning approach for video captioning","venue":null,"work_id":"25d8b141-4451-470f-a4e8-155e67bd02af","year":2021},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.915800Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:7607737922c0db23b3273ae975a2bbe90a598919afc92070fc8e6c44309c7736","observation_id":"d17947a0-38b3-464a-9602-b35d860122c3","resolution":{"observed_at":"2026-08-12T19:48:34.102615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:34.013453Z","title":"Densecap: Fully convolutional localization networks for dense caption- ing","venue":null,"work_id":"e7cb13fd-ea83-43b1-bf13-6b2efa841b4a","year":2016},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.918945Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:d8a89df0cc3614f6e3d376f09b97084c6cbb208ff5e1695505e2b7b1e79de38c","observation_id":"6f1234c7-bf84-4598-9456-bbe1214a6cdc","resolution":{"observed_at":"2026-08-12T19:48:34.017340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:31.922299Z","title":"Panoptic studio: A massively multiview system for social motion capture","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.922299Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:52804ae5ae05c14be25b528071b9f6e152149f07812901c778080d8e73cc6070","observation_id":"d47208de-a2dd-4b7b-8d62-14b5c389e0ac","resolution":{"observed_at":"2026-08-12T19:48:31.922299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1607.01759","last_updated":"2016-08-09T17:38:43Z","snapshot_observed_at":"2026-08-14T01:16:52.740875Z","submitted_at":"2016-07-06T19:40:15Z","title":"Bag of Tricks for Efficient Text Classification","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1607.01759","snapshot_observed_at":"2026-08-12T19:48:31.925830Z","title":"Bag of tricks for efficient text classification","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.925830Z"},"links":{"cited_paper":"/paper/1607.01759","citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:2d460a8b0672ace849e8b8bfc1ee6dc2296e41da28d91065658fd05209f09005","observation_id":"40ae1662-109f-44e9-8dfd-0657fd420ee1","resolution":{"observed_at":"2026-08-12T19:48:31.925830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:31.929993Z","title":"Repurpos- ing diffusion-based image generators for monocular depth estimation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.929993Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:7090ca723042ea4386e670b693bf20ee5fac8528984d1d207e55ff1d0ac0d657","observation_id":"76af935e-dc95-477c-bfae-97f1c819d2e4","resolution":{"observed_at":"2026-08-12T19:48:31.929993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12569","last_updated":"2024-08-27T02:31:42Z","snapshot_observed_at":"2026-08-12T22:59:06.904334Z","submitted_at":"2024-08-22T17:37:27Z","title":"Sapiens: Foundation for Human Vision Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12569","snapshot_observed_at":"2026-08-12T19:48:31.934189Z","title":"Sapiens: Foundation for human vision mod- els","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.934189Z"},"links":{"cited_paper":"/paper/2408.12569","citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:2752e7a527e0f94b89109aedd26f48e445247691d130dcc12abfdf196bff49d4","observation_id":"249e9ee9-a0b9-4c23-aa82-3acd43c39d26","resolution":{"observed_at":"2026-08-12T19:48:31.934189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.992603Z","title":"Sapiens discrete models repository","venue":null,"work_id":"fc7f3568-77d2-4783-8e9a-4253163ec035","year":null},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.938095Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:f0f470ba0dd649a2bb06b307ad6974b20cdbc098a74a962e4844b51fc5864ab1","observation_id":"546e132e-750a-44e6-8050-4b81d1e8c71f","resolution":{"observed_at":"2026-08-12T19:48:33.995550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.928577Z","title":"Sapiens: Foundation for human vision mod- els","venue":null,"work_id":"2d08b382-8637-4249-a3bc-750d1a35ba3e","year":2025},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:31.941568Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:2446639e503e98d8b0d3a5342d74513a3f1203f966845a94de8d6705acd14d08","observation_id":"48382ebc-26ae-4339-8b4f-ec86d9f60d72","resolution":{"observed_at":"2026-08-12T19:48:33.950686Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.002094Z","title":"Segment any- thing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.002094Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:b0018cffd7b94e18225ffe40ea05e4c30004568d0e5be1f18807a4b493dd6929","observation_id":"817d3746-b839-4b86-a153-18c47e87bd58","resolution":{"observed_at":"2026-08-12T19:48:32.002094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.840776Z","title":"Visual genome: 9 Connecting language and vision using crowdsourced dense image annotations","venue":null,"work_id":"5ae19420-2f21-405b-9f02-3700f569db7d","year":2017},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.083097Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:f2064543dcb9875a0cf8ee076c005f8d81a01f299e42bc56646b60c6278a385a","observation_id":"b2f05a6b-b761-45fe-bae2-a4fd173184fa","resolution":{"observed_at":"2026-08-12T19:48:33.844342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.116834Z","title":"Imagenet classification with deep convolutional neural net- works","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.116834Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:4ec171ab2ad6f5d803376f0766d6db5b89d609b890ef3bb0907748ea7c0dd88a","observation_id":"084f3fdd-208b-40d6-bef4-a1f57c08dd64","resolution":{"observed_at":"2026-08-12T19:48:32.116834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.824417Z","title":"Motion-x: A large- scale 3d expressive whole-body human motion dataset","venue":null,"work_id":"3697748e-7c77-4853-b0eb-f881bb267d7d","year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.120634Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:805273399380f577962feb502d40bcd0f095fae26d334a221b65f9339aebabd5","observation_id":"22ba0c89-ac3c-4045-92e5-1622eda7ad7d","resolution":{"observed_at":"2026-08-12T19:48:33.828702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.123328Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.123328Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:232b9c00290a21978869c828360e8c790a2c91e86d2cb3b9eea51bcd730af5d3","observation_id":"526af1bf-e9a0-4af5-9fb1-faa39caf0cbf","resolution":{"observed_at":"2026-08-12T19:48:32.123328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.678711Z","title":"Learning features combination for human action recognition from skeleton sequences","venue":null,"work_id":"ec992233-2d79-4a78-afb0-9029e2d1a211","year":2017},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.126488Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:01bfe17c3e5c9641426251a1a74464b685debd47dc082487cec19cb4dac126c4","observation_id":"a083d145-9c55-4830-b50a-1792b621ce47","resolution":{"observed_at":"2026-08-12T19:48:33.750260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.586405Z","title":"Exploring the limits of weakly supervised pretraining","venue":null,"work_id":"d96dfea0-86ac-4738-a36d-ba040f9f0064","year":2018},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.129923Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:f5ea291277afac535995b1a8ad1bdfa0641cc72522d4766eda3056ef442e3ba9","observation_id":"9f99ddc7-2d21-4889-9c4d-6192ed8f157b","resolution":{"observed_at":"2026-08-12T19:48:33.621617Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.578088Z","title":"Amass: Archive of motion capture as surface shapes","venue":null,"work_id":"28783463-c9a7-47cb-a42d-49f445535b97","year":2019},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.132979Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:04366e07545a16342df321550215637823ddca30ab25cce86350c9b5177fd4c2","observation_id":"0e543e94-e524-41f3-99a0-726ee0ba036a","resolution":{"observed_at":"2026-08-12T19:48:33.581219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.567578Z","title":"Y-net: joint segmen- tation and classification for diagnosis of breast biopsy im- ages","venue":null,"work_id":"c572edaf-120e-4940-986b-cbec27520bf6","year":2018},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.135412Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:02d4380433d7ba1507ab21320fc60a862e8d4d9678404f67ccc27fca0a6f0520","observation_id":"3589cf7e-e5e0-435d-b69e-4e18f3fdc356","resolution":{"observed_at":"2026-08-12T19:48:33.572281Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1301.3781","last_updated":"2013-09-07T00:30:40Z","snapshot_observed_at":"2026-07-06T03:04:11.148340Z","submitted_at":"2013-01-16T18:24:43Z","title":"Efficient Estimation of Word Representations in Vector Space","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1301.3781","snapshot_observed_at":"2026-08-12T19:48:32.138469Z","title":"Efficient estimation of word representa- tions in vector space","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.138469Z"},"links":{"cited_paper":"/paper/1301.3781","citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:9bd6a28a71a7cf3a229a2ca7d6c2530c3ecf164d8abc601c1f3e0234a75ba172","observation_id":"63c33264-156e-456e-9dcf-ca9c2577765f","resolution":{"observed_at":"2026-08-12T19:48:32.138469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.141714Z","title":"Distributed representations of words and phrases and their compositionality","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.141714Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:c6e89d6460061d71bb93f709d97610c3e61083a5ba32bf2062b9f6cc1f159514","observation_id":"809cf054-9d5e-4d2b-a071-92e6e90facbb","resolution":{"observed_at":"2026-08-12T19:48:32.141714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.553774Z","title":"Stacked hour- glass networks for human pose estimation, 2016","venue":null,"work_id":"ccdcbd41-3827-4c73-9692-b5de4f7aadbb","year":2016},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.144662Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:e3e74228b687df277af04b39e6d562e38ce64d2da6d3f38712226b8348a95520","observation_id":"a6125752-1b9b-4dcc-9a44-3001cd5cf3bf","resolution":{"observed_at":"2026-08-12T19:48:33.557585Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.546251Z","title":"Derpa- nis, and Kostas Daniilidis","venue":null,"work_id":"79088b1b-0c50-4671-be86-480254234973","year":2017},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.147584Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:50edac044bd859f0944e5a44d4ac08adef238b56cbc8ce43b03bbbf0fa2b76bf","observation_id":"61c1ff5e-8eea-419b-a526-bf2877e9dd4f","resolution":{"observed_at":"2026-08-12T19:48:33.549075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.537925Z","title":null,"venue":null,"work_id":"992f297f-5a95-47c6-8785-2fa0e8ae77c6","year":2014},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.150508Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:fd1dcecb677a79bffd9cfce0ef3e8f5aaeadb477e3b9c7446105753e30c049de","observation_id":"6ef337c8-2bc3-48b6-b97d-836f9ae66121","resolution":{"observed_at":"2026-08-12T19:48:33.540330Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.154476Z","title":"Unidepth: Universal monocular metric depth estimation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.154476Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:021528f49a0b4a54712e0e9d2b3b3c9ba6418ad11a2bdcb1855110dcb1eddf02","observation_id":"be9da4e2-8603-4bf8-a7f3-db0af0dfea04","resolution":{"observed_at":"2026-08-12T19:48:32.154476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.157855Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.157855Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:6e9cf85a33bf694607bcb837ccb67d83b4b6db56962bf946c73a7f59eefc4395","observation_id":"2086889f-b333-4a04-9184-1f0cfdfa70de","resolution":{"observed_at":"2026-08-12T19:48:32.157855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.516855Z","title":"Towards robust monocular depth estimation: Mixing datasets for zero-shot cross-dataset transfer","venue":null,"work_id":"125d9078-1e6a-46ae-b3ea-b450331d7503","year":2020},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.263913Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:4f2d23ea6835de6ec348abeb88346b1159321c7d99d944a6a05fee7733c48645","observation_id":"0c00a0a4-438d-4c2d-9180-935209b8ad35","resolution":{"observed_at":"2026-08-12T19:48:33.520879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.345416Z","title":"You only look once: Unified, real-time object de- tection","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.345416Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:71d86f4767c0d44d35180558dbd699efa084ad9bbfc190e1877a6aef9497ce0b","observation_id":"78ec2cd0-4c57-4c13-a2e5-bcedd7bdea3e","resolution":{"observed_at":"2026-08-12T19:48:32.345416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.349694Z","title":"Faster r-cnn: Towards real-time object detection with region proposal networks","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.349694Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:bb9bdd321dc9f75bc269dfae695c6f471b5662db171313eb86fa04a5c61995ef","observation_id":"3980dc14-b5a7-4a81-924a-02abb9a15e68","resolution":{"observed_at":"2026-08-12T19:48:32.349694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.366197Z","title":"Lcr-net: Localization-classification-regression for human pose","venue":null,"work_id":"9a726fd2-b722-40ea-bfc2-366357662055","year":2017},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.353152Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:d8997f316f0e65c311793918b4e0c2a5b789431e47305561d82098927e1808a3","observation_id":"1aadc054-d7ff-48af-b159-4cdd678cb4db","resolution":{"observed_at":"2026-08-12T19:48:33.441396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.356258Z","title":"U- net: Convolutional networks for biomedical image segmen- tation","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.356258Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:65f76251fb1aa0e993ce294f008a1f1f6084f8d236ed479262f600d136caa14e","observation_id":"78cece2a-b6ed-4ffe-a3d7-cfddf65b5900","resolution":{"observed_at":"2026-08-12T19:48:32.356258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.359234Z","title":"Laion-5b: An open large-scale dataset for training next generation image-text models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.359234Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:0dec0fd6858f7a2436a6806f06772ca3e0f90d587eb2cc82df856c4cea717423","observation_id":"0d9cfcae-7323-4eb8-a2ec-739f650676d3","resolution":{"observed_at":"2026-08-12T19:48:32.359234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1409.1556","last_updated":"2015-04-10T16:25:04Z","snapshot_observed_at":"2026-08-14T04:13:35.056640Z","submitted_at":"2014-09-04T19:48:04Z","title":"Very Deep Convolutional Networks for Large-Scale Image Recognition","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1409.1556","snapshot_observed_at":"2026-08-12T19:48:32.362357Z","title":"Very deep convo- lutional networks for large-scale image recognition","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.362357Z"},"links":{"cited_paper":"/paper/1409.1556","citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:b3a4d7ae7f1e6bb9eeaa21bf3fc0536794c14b75d82008d759f6b68eef12d0ed","observation_id":"fb2359f6-b880-46b6-8b69-6fc8656090fe","resolution":{"observed_at":"2026-08-12T19:48:32.362357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.365988Z","title":"Revisiting unreasonable effectiveness of data in deep learning era","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.365988Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:48ae65f08f251e08b8ab92b8978a42ca9c7dbac388765a670a34834472ffee09","observation_id":"aa671b21-aa1b-437b-9dde-05c9a0c9a442","resolution":{"observed_at":"2026-08-12T19:48:32.365988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.343051Z","title":"Mask-yolo","venue":null,"work_id":"992b7b3f-5502-4775-993a-8448f3753723","year":2018},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.369181Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:74539b8dfde4a2584e118c5dcfd9d15799847725f27a70346bfc678123265009","observation_id":"48d3e48a-0dd9-46b8-8b38-87ed0ecbc87d","resolution":{"observed_at":"2026-08-12T19:48:33.346046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.372784Z","title":"Deep high-resolution representation learning for human pose es- timation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.372784Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:26d83c7e8148d4ce069a226ad920994b5016e79a4059392bc4676df363864d95","observation_id":"bfe92ecd-e62f-48ec-bd44-92f29778b4d2","resolution":{"observed_at":"2026-08-12T19:48:32.372784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.426648Z","title":"Going deeper with convolutions","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.426648Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:b2744ec3dabeeb6f8e7b26a9df5626b92bef6a70046a530e65824f3ddee7bc28","observation_id":"d1cd060e-49eb-4ff1-bee3-d65e31d2f793","resolution":{"observed_at":"2026-08-12T19:48:32.426648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.324162Z","title":"Joint training of a convolutional network and a graphical model for human pose estimation","venue":null,"work_id":"ee7080c6-2fef-4cd0-ab95-4e8782a89aea","year":2014},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.490731Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:44e3bcf0167b026df50ff8856223417ef954a30201ce2d1a4d6040f0e3bb1cd8","observation_id":"59ea1e12-48d8-47aa-acf9-f7a9673e7583","resolution":{"observed_at":"2026-08-12T19:48:33.327248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.185269Z","title":"Deeppose: Human pose estimation via deep neural networks","venue":null,"work_id":"00266dc4-8305-44b1-8609-5a1d80b34258","year":2014},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.494573Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:2c36ad38c4cd258366e79923efe87324eb101e5c8a060e3a4c79d4e90d9c1e33","observation_id":"341623ba-bfa1-47d8-9676-cff011fdb21b","resolution":{"observed_at":"2026-08-12T19:48:33.295813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.498323Z","title":"Attention is all you need","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.498323Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:b54354b087b957b1119a76c1ee449fcb58055c25d1506012dcd2370261ce641a","observation_id":"2026cb34-3bec-4e5f-bf61-a4792a93ede3","resolution":{"observed_at":"2026-08-12T19:48:32.498323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.501928Z","title":"Deep high-resolution repre- sentation learning for visual recognition","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.501928Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:0fcf172d5d22e0a606a1f3382db254b606398840883b4d96e8a1347f9398011b","observation_id":"3c14e94f-9f80-4137-b377-7a40cc9bf495","resolution":{"observed_at":"2026-08-12T19:48:32.501928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.166769Z","title":"Y-net: a one-to-two deep learning framework for digital holographic reconstruction","venue":null,"work_id":"f22f8cec-099e-462d-891a-593f7e31fe32","year":2019},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.504928Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:d4b6887b8e69cd0d55894cbbfa656fcc6165774ac8abd7eaeffd81ebb4877395","observation_id":"7fa5486b-09f6-49d6-9255-b8ef1b6333d2","resolution":{"observed_at":"2026-08-12T19:48:33.170351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.508802Z","title":"Dust3r: Geometric 3d vi- sion made easy","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.508802Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:547817176cb19a55c882076a9ad0f6e0185262e748fbabf0bb8f10df9a323779","observation_id":"2293a741-20c4-470b-a5ae-451939fc08fb","resolution":{"observed_at":"2026-08-12T19:48:32.508802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.055570Z","title":"Detectron2","venue":null,"work_id":"5915a26b-ddaa-4ab8-b36e-29b37fb19d5d","year":2019},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.511448Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:7867e63a475bfd43f0b2cf84c6020eeea0d1a55e9f640aa3f2eb6dcef4615b77","observation_id":"1d1cda28-5ac5-4b04-8792-b9b17cead7c6","resolution":{"observed_at":"2026-08-12T19:48:33.155649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1505.00853","last_updated":"2015-11-27T06:58:14Z","snapshot_observed_at":"2026-07-06T04:16:54.272660Z","submitted_at":"2015-05-05T01:16:39Z","title":"Empirical Evaluation of Rectified Activations in Convolutional Network","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1505.00853","snapshot_observed_at":"2026-08-12T19:48:32.515471Z","title":"Empirical evaluation of rectified activations in con- volutional network","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.515471Z"},"links":{"cited_paper":"/paper/1505.00853","citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:79fa461a90e96b5775e3b210eae40c09dcf70ea3a11900e7e9dd7f2be900d461","observation_id":"8950cc69-942a-4279-803d-5428e95bd62f","resolution":{"observed_at":"2026-08-12T19:48:32.515471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.046145Z","title":"Unifying flow, stereo and depth estimation","venue":null,"work_id":"f6097ab4-4ddb-46ed-833e-68ed95f41e7e","year":2023},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.519175Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:4cc00f5317c59cc60904b3ccf7384d0f0603622aaec000db70e4f1c30adb0280","observation_id":"16e4fbf8-a0e3-4195-a75b-374508f50223","resolution":{"observed_at":"2026-08-12T19:48:33.049507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.035062Z","title":"Zoomnas: search- ing for whole-body human pose estimation in the wild.IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(4):5296–5313, 2022","venue":null,"work_id":"a75acf45-a5b5-41ca-aa8c-e897c1f1171b","year":2022},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.522539Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:4f50ff348832ac73df3f695b4dddaf81e908e628d82daa4dd15389774ebe57c8","observation_id":"753987b0-d2fa-4ab8-aaa3-28fd3c6f4e0e","resolution":{"observed_at":"2026-08-12T19:48:33.039696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.025232Z","title":"Depth anything: Unleashing the power of large-scale unlabeled data","venue":null,"work_id":"4eb9f86a-d628-4e70-b5dc-2f1fc052c4a2","year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.525915Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:4af10f6a89cd4f5e57fa64bf0a77f91bfec563bee6f7b84b1a206597b67e6192","observation_id":"a241ed22-71e7-4d35-b08b-1f6dde22ba20","resolution":{"observed_at":"2026-08-12T19:48:33.029033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09414","last_updated":"2024-10-20T11:24:09Z","snapshot_observed_at":"2026-07-06T18:30:32.982860Z","submitted_at":"2024-06-13T17:59:56Z","title":"Depth Anything V2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09414","snapshot_observed_at":"2026-08-12T19:48:32.530018Z","title":"Depth any- thing v2","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.530018Z"},"links":{"cited_paper":"/paper/2406.09414","citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:824ca42a0b0fee1fd0b0caff4e359d4b0bb70de1f8de84e4dab3940735cfd405","observation_id":"c1354a33-3ee3-4967-b85e-0b0ca0fc6370","resolution":{"observed_at":"2026-08-12T19:48:32.530018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:33.012270Z","title":"End-to-end learning of deformable mixture of parts and deep convolutional neural networks for human pose esti- mation","venue":null,"work_id":"2109cd67-0029-4125-b359-6c8c75c4122c","year":2016},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.533597Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:3f0f668a9cc0e37db4535d621bbc697326879d62ff485fb22b611190a0b7859e","observation_id":"c34cc9b5-2ad8-4723-9b2c-823938e1acc8","resolution":{"observed_at":"2026-08-12T19:48:33.016587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.999395Z","title":"Dptext-detr: Towards better scene text detection with dynamic points in transformer","venue":null,"work_id":"941c6587-3534-45d2-9819-6ce14a7af59e","year":2023},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.577494Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:9c2163b44e7f90b5748a90f3bf0ca49dba00a2d9d31737f77d60223f7bb75cc5","observation_id":"34546536-b324-400e-a137-5f4da12cf965","resolution":{"observed_at":"2026-08-12T19:48:33.003423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.986570Z","title":"Differential transformer, 2024","venue":null,"work_id":"08436465-45a1-4fa9-8ddb-e00bc3624ca9","year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.659966Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:d800434597eeacf9d987025ccbfe8cba7d818675a4ac650a9c446ae9d40a6b93","observation_id":"fc42eb17-c79f-4c14-8074-0ec034090818","resolution":{"observed_at":"2026-08-12T19:48:32.990798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.699287Z","title":"Metric3d: Towards zero-shot metric 3d prediction from a single image","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.699287Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:ec88a83628ebcd0c3f4197c3dac73dc5df7b6b29a0ed9a8a0d7889d52b60e60c","observation_id":"daf28c49-891e-4fc3-955e-76127723f4e4","resolution":{"observed_at":"2026-08-12T19:48:32.699287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.939882Z","title":"Learn- ing from multiple teacher networks","venue":null,"work_id":"66377d5b-6a87-4465-a553-488636efe75c","year":2017},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.702710Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:35f363d24f5248364be56c6bef521d223d1a67664b9ce14dbb52d15079381448","observation_id":"bdc3a0ef-dd0d-4440-bd1c-9e029e715f33","resolution":{"observed_at":"2026-08-12T19:48:32.971313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:48:32.822688Z","title":"Lite-hrnet: A lightweight high-resolution network","venue":null,"work_id":"4cd53c19-afcf-4aaf-883c-26daa887ca81","year":2021},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.705719Z"},"links":{"citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:da1a28efd4c08feffe6987b82440fa0aa2117702ea61a156723c5a52f7854daf","observation_id":"ce5ed699-a2cf-45ac-b305-b824d190de42","resolution":{"observed_at":"2026-08-12T19:48:32.879333Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.03752","last_updated":"2024-12-23T17:57:29Z","snapshot_observed_at":"2026-08-12T22:50:28.171909Z","submitted_at":"2024-09-05T17:59:12Z","title":"Attention Heads of Large Language Models: A Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.03752","snapshot_observed_at":"2026-08-12T19:48:32.707991Z","title":"Attention heads of large language models: A survey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-12T19:48:32.707991Z"},"links":{"cited_paper":"/paper/2409.03752","citing_paper":"/paper/2411.10334"},"observation_digest":"sha256:07dc17cd9cc762f52ea6349df7056878e17e59abe36553b6590e272b14132dd0","observation_id":"bd8d7780-775a-429b-ad41-918618667f47","resolution":{"observed_at":"2026-08-12T19:48:32.707991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2411.10334","last_updated":"2024-11-15T16:33:59Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T14:07:35.708108Z","submitted_at":"2024-11-15T16:33:59Z","title":"Y-MAP-Net: Real-time depth, normals, segmentation, multi-label captioning and 2D human pose in RGB images"},"reference_resolution":{"displayed":76,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":40,"verified_exact":0,"verified_fuzzy":36},"total_outbound_references":76},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 76 of 76 outbound references and 0 inbound Pith citation observations for arXiv:2411.10334."}