{"work":{"id":"7083a41e-5666-435b-ab26-c753f6490b9a","openalex_id":"https://openalex.org/W3174429308","doi":"10.1109/cvpr46437.2021.00033","arxiv_id":"6437.2021","raw_key":null,"title":"P., Huang, Z., Romero, A","authors":null,"authors_text":"d’Apolito, S","year":2021,"venue":null,"abstract":null,"external_url":"https://arxiv.org/abs/6437.2021","cited_by_count":16,"metadata_source":"arxiv_reference","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":null,"created_at":"2026-05-09T18:55:07.570959+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":false,"display_title":"Derf: Decomposed radiance fields","render_title":"Derf: Decomposed radiance fields"},"hub":{"state":{"work_id":"7083a41e-5666-435b-ab26-c753f6490b9a","tier":"super_hub","tier_reason":"100+ Pith inbound or 10,000+ external citations","pith_inbound_count":164,"external_cited_by_count":16,"distinct_field_count":17,"first_pith_cited_at":"2022-07-05T07:17:43+00:00","last_pith_cited_at":"2026-07-09T14:29:31+00:00","author_build_status":"needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-23T09:39:21.177775+00:00","tier_text":"super_hub"},"tier":"super_hub","role_counts":[{"context_role":"background","n":24},{"context_role":"dataset","n":5},{"context_role":"method","n":5},{"context_role":"baseline","n":3}],"polarity_counts":[{"context_polarity":"background","n":25},{"context_polarity":"use_dataset","n":5},{"context_polarity":"baseline","n":3},{"context_polarity":"use_method","n":3},{"context_polarity":"unclear","n":1}],"runs":{"ask_index":{"job_type":"ask_index","status":"succeeded","result":{"title":"Derf: Decomposed radiance fields","claims":[{"claim_text":"archical cognition through nested divide-and-conquer pro- cessing, and transcribes refined holistic representations into recognition model parameters to enhance sensitivity across the entire object. • We demonstrate that modeling holistic cues substantially improves the discrimination of highly similar subcategories, with DHCNet achieving a +4.2% average accuracy gain over state-of-the-art methods [12] across five large-scale Ultra- FGVC benchmarks. 2 Related Work Ultra-fine-grained visual categ","claim_type":"baseline","confidence":0.95,"evidence_strength":"citation_context"},{"claim_text":"Method Abs Rel↓Sq Rel↓RMSE↓log RMSE↓ δ <1.25↑δ <1.25 2 ↑δ <1.25 3 ↑ VNL [61] 0.108−0.416− 0.875 0.976 0.994 DA V [24] 0.108−0.412− 0.882 0.980 0.996 DPT* [40] 0.110−0.357− 0.904 0.988 0.998 TransDepth [58] 0.106−0.365− 0.900 0.983 0.996 ASN [34] 0.101−0.377− 0.890 0.982− PackNet-SAN* [20] 0.106−0.393− 0.892 0.979 0.995 PW A [29] 0.105−0.374− 0.892 0.985 0.997 AdaBins [4] 0.103−0.364− 0.903 0.984 0.997 LocalBins [5] 0.099−0.357− 0.907 0.987 0.998 BinsFormer [30] 0.094−0.330− 0.925 0.989 0.997 P3D","claim_type":"baseline","confidence":0.95,"evidence_strength":"citation_context"},{"claim_text":"Video (R) [76], Top&Random (R) [77] ,Reddit Images (R) [78], UIV (C) [74] Emotions and Social Signals Affective Analysis Pitts Ads Dataset (C) [64], Video Emotion Dataset (C) [11], Ekman Emotion Dataset (C) [11], VAAD (C) [79], iMiGUE (C) [80], EALD (Q) [81], VCE (C) [82], V2V (R) [82], VEATIC (R) [83], MERR (C,Cap) [14], 3MASSIV (C) [70], LAMBDA (Q) [63], ArtEmis (C,Cap) [84], EmoSet (C) [85] Relationships SRIV (C) [86], ViSR (C) [87], PERR (C) [88], MovieGraphs (Q) [89], LVU (C) [66], VideoAds","claim_type":"dataset","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"(C) [7], SemArt (Ret) [73], UIV (C) [74] User Behaviour Modeling/Virality LVU (R) [66], MicroVideos (R) [75], CMU Viral Video (R) [76], Top&Random (R) [77] ,Reddit Images (R) [78], UIV (C) [74] Emotions and Social Signals Affective Analysis Pitts Ads Dataset (C) [64], Video Emotion Dataset (C) [11], Ekman Emotion Dataset (C) [11], VAAD (C) [79], iMiGUE (C) [80], EALD (Q) [81], VCE (C) [82], V2V (R) [82], VEATIC (R) [83], MERR (C,Cap) [14], 3MASSIV (C) [70], LAMBDA (Q) [63], ArtEmis (C,Cap) [84],","claim_type":"dataset","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"Intent Oops (C) [48], IntentQA (Q) [49], FunQA (Q) [50], Vid2Int (C) [51], MIntRec2.0 (C) [52], VCR (Q) [53], Intentonomy (C) [54] Visual Aesthetics KoNViD-1k (R) [55], LSVQ (R) [56], DIVIDE- 3k (R) [57], AVA (C) [58], LIVE-itW (R) [59], MDID (C) [60], KonIQ-10k (R) [61], SPAQ (R) [62], LAMBDA (Q) [63] Semantic Theme Understanding Pitts Ads Dataset (C) [64], YouTube-8M (C) [65], LVU (C) [66], Tencent AVS (C) [67], MM-AU (C) [68], VideoAds [69], 3MASSIV (C) [70], LAION- 400M (Ret) [71], DEEPEVAL ","claim_type":"dataset","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"T able 1 Datasets and benchmarks organized in abstract concept recognition subdomains and colour coded (video, image). Dataset type legend: (C) Classification, (Q) QnA, (R) Regression, (Cap) Captioning, (Ret) Retrieval, (Ra) Ranking. Category Subcategory Datasets/Benchmarks Perception Understanding Intent Oops (C) [48], IntentQA (Q) [49], FunQA (Q) [50], Vid2Int (C) [51], MIntRec2.0 (C) [52], VCR (Q) [53], Intentonomy (C) [54] Visual Aesthetics KoNViD-1k (R) [55], LSVQ (R) [56], DIVIDE- 3k (R) [","claim_type":"dataset","confidence":0.9,"evidence_strength":"citation_context"}],"why_cited":"Pith tracks Derf: Decomposed radiance fields because it crossed a citation-hub threshold. Current citing contexts most often use it as background evidence (23 contexts).","role_counts":[{"n":23,"context_role":"background"},{"n":5,"context_role":"dataset"},{"n":5,"context_role":"method"},{"n":3,"context_role":"baseline"}]},"error":null,"updated_at":"2026-06-26T11:35:01.200092+00:00"},"author_expand":{"job_type":"author_expand","status":"succeeded","result":{"authors_linked":[{"id":"d03ff47d-e99d-4eb9-94a9-b099464a4283","orcid":null,"display_name":"Guang Feng"},{"id":"cd6d893d-0999-4bcb-a1ca-4326a55c48b7","orcid":null,"display_name":"Zhiwei Hu"},{"id":"e30ac4d9-e1bc-43ac-8061-b58183783da0","orcid":null,"display_name":"Lihe Zhang"},{"id":"29fc5c12-582c-4021-9869-70192da286ca","orcid":null,"display_name":"and Huchuan Lu"}]},"error":null,"updated_at":"2026-06-26T11:35:01.192963+00:00"},"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-14T15:41:59.292882+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"Masked autoencoders are scalable vision learners","work_id":"0a23d1b7-bd56-43cc-8a80-7c43ce994e1e","shared_citers":14},{"title":"& Vondrick, C","work_id":"b8a8bb9e-1d31-40e2-9cab-ae21e338dde6","shared_citers":13},{"title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","work_id":"7efbc2dd-b0f2-4f71-bb1c-d2fcf110d805","shared_citers":11},{"title":"IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , year =","work_id":"9da51225-b7bd-4032-b7db-ca577971dafe","shared_citers":11},{"title":"In: 2023 IEEE/CVF Conference on Com- puter Vision and Pattern Recognition (CVPR)","work_id":"b9701eca-d05e-4d2e-9045-6761df4ba175","shared_citers":11},{"title":"Editing conditional radiance fields","work_id":"3820f598-11b0-45c3-8c99-0079181ac0a7","shared_citers":9},{"title":"MambaVision: A hybrid Mamba- Transformer vision backbone","work_id":"d0e5199d-8907-47b1-905a-07ab8b623a4c","shared_citers":6},{"title":"Tomasi and R","work_id":"135418b1-cafd-49fd-803d-1ca6433d4b1b","shared_citers":6},{"title":"URLhttp://dx.doi.org/10.1109/CVPR.2016.90","work_id":"b353bda2-591d-479a-9c8b-22dfcba12431","shared_citers":5},{"title":"Bovik, H.R","work_id":"a9ba8a9e-c00a-45ff-9d73-a0ae0919d283","shared_citers":4},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":4},{"title":"why should I trust you?","work_id":"238df2e4-a3e5-46f3-860e-3ae2b0094b97","shared_citers":4},{"title":"In: IEEE/CVF Winter Conference on Applications of Computer Vision, WACV 2023, Waikoloa, HI, USA, January 2-7","work_id":"f85bf8f6-efad-46de-bf88-1ca18d39b5c7","shared_citers":3},{"title":"In: Proceedings of the IEEE/CVF International Conference on Computer 16 J","work_id":"3d71525d-5fe4-4476-9dd0-dd2c0eef947e","shared_citers":3},{"title":"Recognizing indoor scenes","work_id":"45b0bfd8-65dc-4252-b2ab-2f6b411d04d0","shared_citers":3},{"title":"2019.01.103","work_id":"a31c4c01-2c91-49a9-88e4-6b1cdb299db0","shared_citers":2},{"title":"Adam: A Method for Stochastic Optimization","work_id":"1910796d-9b52-4683-bf5c-de9632c1028b","shared_citers":2},{"title":"Aggregated Residual Transformations for Deep Neural Networks","work_id":"0df5c3b6-667a-4f46-a400-d799abe1cfc3","shared_citers":2},{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":2},{"title":"arXiv2502.06608(2025) 5, 6, 10","work_id":"bf744acd-a6e4-4ba7-98eb-43811d95fa81","shared_citers":2},{"title":"arXiv2506.15442(2025) 10","work_id":"ee52f4d7-462f-4491-9549-4160820ae563","shared_citers":2},{"title":"arXiv preprint arXiv:1711.04623 , year=","work_id":"bcb52129-9115-4696-8926-64f96214f2d7","shared_citers":2},{"title":"B., He, K., & Doll \\' a r, P","work_id":"01d423c6-2d6b-4f3a-979f-9eb323635cfa","shared_citers":2},{"title":"Brendan and Mironov, Ilya and Talwar, Kunal and Zhang, Li , year=","work_id":"518424d2-1085-4f06-9f9f-1c3aa7913ecb","shared_citers":2}],"time_series":[{"n":1,"year":2023},{"n":1,"year":2024},{"n":1,"year":2025},{"n":43,"year":2026}],"dependency_candidates":[]},"error":null,"updated_at":"2026-05-14T15:41:59.314703+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"items":[{"title":"Qwen3 Technical Report","outcome":"unchanged","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"counts":{"fixed":0,"merged":0,"unchanged":1,"quarantined":0,"needs_external_resolution":0},"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-14T15:42:10.123375+00:00"},"role_polarity":{"job_type":"role_polarity","status":"succeeded","result":{"title":"Derf: Decomposed radiance fields","claims":[{"claim_text":"archical cognition through nested divide-and-conquer pro- cessing, and transcribes refined holistic representations into recognition model parameters to enhance sensitivity across the entire object. • We demonstrate that modeling holistic cues substantially improves the discrimination of highly similar subcategories, with DHCNet achieving a +4.2% average accuracy gain over state-of-the-art methods [12] across five large-scale Ultra- FGVC benchmarks. 2 Related Work Ultra-fine-grained visual categ","claim_type":"baseline","confidence":0.95,"evidence_strength":"citation_context"},{"claim_text":"Method Abs Rel↓Sq Rel↓RMSE↓log RMSE↓ δ <1.25↑δ <1.25 2 ↑δ <1.25 3 ↑ VNL [61] 0.108−0.416− 0.875 0.976 0.994 DA V [24] 0.108−0.412− 0.882 0.980 0.996 DPT* [40] 0.110−0.357− 0.904 0.988 0.998 TransDepth [58] 0.106−0.365− 0.900 0.983 0.996 ASN [34] 0.101−0.377− 0.890 0.982− PackNet-SAN* [20] 0.106−0.393− 0.892 0.979 0.995 PW A [29] 0.105−0.374− 0.892 0.985 0.997 AdaBins [4] 0.103−0.364− 0.903 0.984 0.997 LocalBins [5] 0.099−0.357− 0.907 0.987 0.998 BinsFormer [30] 0.094−0.330− 0.925 0.989 0.997 P3D","claim_type":"baseline","confidence":0.95,"evidence_strength":"citation_context"},{"claim_text":"Video (R) [76], Top&Random (R) [77] ,Reddit Images (R) [78], UIV (C) [74] Emotions and Social Signals Affective Analysis Pitts Ads Dataset (C) [64], Video Emotion Dataset (C) [11], Ekman Emotion Dataset (C) [11], VAAD (C) [79], iMiGUE (C) [80], EALD (Q) [81], VCE (C) [82], V2V (R) [82], VEATIC (R) [83], MERR (C,Cap) [14], 3MASSIV (C) [70], LAMBDA (Q) [63], ArtEmis (C,Cap) [84], EmoSet (C) [85] Relationships SRIV (C) [86], ViSR (C) [87], PERR (C) [88], MovieGraphs (Q) [89], LVU (C) [66], VideoAds","claim_type":"dataset","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"(C) [7], SemArt (Ret) [73], UIV (C) [74] User Behaviour Modeling/Virality LVU (R) [66], MicroVideos (R) [75], CMU Viral Video (R) [76], Top&Random (R) [77] ,Reddit Images (R) [78], UIV (C) [74] Emotions and Social Signals Affective Analysis Pitts Ads Dataset (C) [64], Video Emotion Dataset (C) [11], Ekman Emotion Dataset (C) [11], VAAD (C) [79], iMiGUE (C) [80], EALD (Q) [81], VCE (C) [82], V2V (R) [82], VEATIC (R) [83], MERR (C,Cap) [14], 3MASSIV (C) [70], LAMBDA (Q) [63], ArtEmis (C,Cap) [84],","claim_type":"dataset","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"Intent Oops (C) [48], IntentQA (Q) [49], FunQA (Q) [50], Vid2Int (C) [51], MIntRec2.0 (C) [52], VCR (Q) [53], Intentonomy (C) [54] Visual Aesthetics KoNViD-1k (R) [55], LSVQ (R) [56], DIVIDE- 3k (R) [57], AVA (C) [58], LIVE-itW (R) [59], MDID (C) [60], KonIQ-10k (R) [61], SPAQ (R) [62], LAMBDA (Q) [63] Semantic Theme Understanding Pitts Ads Dataset (C) [64], YouTube-8M (C) [65], LVU (C) [66], Tencent AVS (C) [67], MM-AU (C) [68], VideoAds [69], 3MASSIV (C) [70], LAION- 400M (Ret) [71], DEEPEVAL ","claim_type":"dataset","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"T able 1 Datasets and benchmarks organized in abstract concept recognition subdomains and colour coded (video, image). Dataset type legend: (C) Classification, (Q) QnA, (R) Regression, (Cap) Captioning, (Ret) Retrieval, (Ra) Ranking. Category Subcategory Datasets/Benchmarks Perception Understanding Intent Oops (C) [48], IntentQA (Q) [49], FunQA (Q) [50], Vid2Int (C) [51], MIntRec2.0 (C) [52], VCR (Q) [53], Intentonomy (C) [54] Visual Aesthetics KoNViD-1k (R) [55], LSVQ (R) [56], DIVIDE- 3k (R) [","claim_type":"dataset","confidence":0.9,"evidence_strength":"citation_context"}],"why_cited":"Pith tracks Derf: Decomposed radiance fields because it crossed a citation-hub threshold. Current citing contexts most often use it as background evidence (23 contexts).","role_counts":[{"n":23,"context_role":"background"},{"n":5,"context_role":"dataset"},{"n":5,"context_role":"method"},{"n":3,"context_role":"baseline"}]},"error":null,"updated_at":"2026-06-26T11:35:01.197548+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"In: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","claims":[],"why_cited":"Pith tracks In: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-14T15:41:57.501649+00:00"}},"summary":{"title":"In: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","claims":[],"why_cited":"Pith tracks In: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"Masked autoencoders are scalable vision learners","work_id":"0a23d1b7-bd56-43cc-8a80-7c43ce994e1e","shared_citers":14},{"title":"& Vondrick, C","work_id":"b8a8bb9e-1d31-40e2-9cab-ae21e338dde6","shared_citers":13},{"title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","work_id":"7efbc2dd-b0f2-4f71-bb1c-d2fcf110d805","shared_citers":11},{"title":"IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , year =","work_id":"9da51225-b7bd-4032-b7db-ca577971dafe","shared_citers":11},{"title":"In: 2023 IEEE/CVF Conference on Com- puter Vision and Pattern Recognition (CVPR)","work_id":"b9701eca-d05e-4d2e-9045-6761df4ba175","shared_citers":11},{"title":"Editing conditional radiance fields","work_id":"3820f598-11b0-45c3-8c99-0079181ac0a7","shared_citers":9},{"title":"MambaVision: A hybrid Mamba- Transformer vision backbone","work_id":"d0e5199d-8907-47b1-905a-07ab8b623a4c","shared_citers":6},{"title":"Tomasi and R","work_id":"135418b1-cafd-49fd-803d-1ca6433d4b1b","shared_citers":6},{"title":"URLhttp://dx.doi.org/10.1109/CVPR.2016.90","work_id":"b353bda2-591d-479a-9c8b-22dfcba12431","shared_citers":5},{"title":"Bovik, H.R","work_id":"a9ba8a9e-c00a-45ff-9d73-a0ae0919d283","shared_citers":4},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":4},{"title":"why should I trust you?","work_id":"238df2e4-a3e5-46f3-860e-3ae2b0094b97","shared_citers":4},{"title":"In: IEEE/CVF Winter Conference on Applications of Computer Vision, WACV 2023, Waikoloa, HI, USA, January 2-7","work_id":"f85bf8f6-efad-46de-bf88-1ca18d39b5c7","shared_citers":3},{"title":"In: Proceedings of the IEEE/CVF International Conference on Computer 16 J","work_id":"3d71525d-5fe4-4476-9dd0-dd2c0eef947e","shared_citers":3},{"title":"Recognizing indoor scenes","work_id":"45b0bfd8-65dc-4252-b2ab-2f6b411d04d0","shared_citers":3},{"title":"2019.01.103","work_id":"a31c4c01-2c91-49a9-88e4-6b1cdb299db0","shared_citers":2},{"title":"Adam: A Method for Stochastic Optimization","work_id":"1910796d-9b52-4683-bf5c-de9632c1028b","shared_citers":2},{"title":"Aggregated Residual Transformations for Deep Neural Networks","work_id":"0df5c3b6-667a-4f46-a400-d799abe1cfc3","shared_citers":2},{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":2},{"title":"arXiv2502.06608(2025) 5, 6, 10","work_id":"bf744acd-a6e4-4ba7-98eb-43811d95fa81","shared_citers":2},{"title":"arXiv2506.15442(2025) 10","work_id":"ee52f4d7-462f-4491-9549-4160820ae563","shared_citers":2},{"title":"arXiv preprint arXiv:1711.04623 , year=","work_id":"bcb52129-9115-4696-8926-64f96214f2d7","shared_citers":2},{"title":"B., He, K., & Doll \\' a r, P","work_id":"01d423c6-2d6b-4f3a-979f-9eb323635cfa","shared_citers":2},{"title":"Brendan and Mironov, Ilya and Talwar, Kunal and Zhang, Li , year=","work_id":"518424d2-1085-4f06-9f9f-1c3aa7913ecb","shared_citers":2}],"time_series":[{"n":1,"year":2023},{"n":1,"year":2024},{"n":1,"year":2025},{"n":43,"year":2026}],"dependency_candidates":[]},"authors":[{"id":"29fc5c12-582c-4021-9869-70192da286ca","orcid":null,"display_name":"and Huchuan Lu","source":"manual","import_confidence":0.72},{"id":"d03ff47d-e99d-4eb9-94a9-b099464a4283","orcid":null,"display_name":"Guang Feng","source":"manual","import_confidence":0.72},{"id":"e30ac4d9-e1bc-43ac-8061-b58183783da0","orcid":null,"display_name":"Lihe Zhang","source":"manual","import_confidence":0.72},{"id":"cd6d893d-0999-4bcb-a1ca-4326a55c48b7","orcid":null,"display_name":"Zhiwei Hu","source":"manual","import_confidence":0.72}]}}