{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:PARVOTX3UNYT3FS6OJIDCFXXII","short_pith_number":"pith:PARVOTX3","schema_version":"1.0","canonical_sha256":"7823574efba3713d965e72503116f7421197df57f4130523ed5e48f6bbb60e24","source":{"kind":"arxiv","id":"2607.16012","version":1},"attestation_state":"computed","paper":{"title":"DPNeXt: A Lightweight Multi-Scale Feature Fusion Framework for Efficient ViT-Based Multi-Task Dense Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CV","authors_text":"David Hyunchul Shim, Jehun Kang, Jungha Wang, Youngjun Hwang","submitted_at":"2026-07-17T14:46:39Z","abstract_excerpt":"Multi-Task Learning (MTL) in robotics perception systems supports comprehensive 3D spatial scene understanding by integrating semantic segmentation and depth estimation. While Vision Foundation Models (VFMs) are increasingly adopted as robust feature encoders, existing decoding strategies present a critical bottleneck. To address this, we propose DPNeXt, a streamlined multi-scale feature fusion decoder and efficient alternative to the standard Dense Prediction Transformer (DPT). DPNeXt uses dual depthwise separable inverted bottlenecks to improve frozen VFM utilization through fusion-centric d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.16012","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-07-17T14:46:39Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"87699fa656a62ef689b64fc96392d7a596369b54fda207e5f9f1bfd7c31cea6e","abstract_canon_sha256":"611fca8cdfa2dba185ec0e06390441f723bbc44f58c1589125136982138d254d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-20T02:19:26.172854Z","signature_b64":"MMGhiNUqVBP0kLZb9whOUffJ0E0tP5j8xScpejUgKze/CkEp73VYfc8gCoobRcJFdn5MAc9hpnUg9FS5IVVpBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7823574efba3713d965e72503116f7421197df57f4130523ed5e48f6bbb60e24","last_reissued_at":"2026-07-20T02:19:26.171977Z","signature_status":"signed_v1","first_computed_at":"2026-07-20T02:19:26.171977Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DPNeXt: A Lightweight Multi-Scale Feature Fusion Framework for Efficient ViT-Based Multi-Task Dense Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CV","authors_text":"David Hyunchul Shim, Jehun Kang, Jungha Wang, Youngjun Hwang","submitted_at":"2026-07-17T14:46:39Z","abstract_excerpt":"Multi-Task Learning (MTL) in robotics perception systems supports comprehensive 3D spatial scene understanding by integrating semantic segmentation and depth estimation. While Vision Foundation Models (VFMs) are increasingly adopted as robust feature encoders, existing decoding strategies present a critical bottleneck. To address this, we propose DPNeXt, a streamlined multi-scale feature fusion decoder and efficient alternative to the standard Dense Prediction Transformer (DPT). DPNeXt uses dual depthwise separable inverted bottlenecks to improve frozen VFM utilization through fusion-centric d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.16012","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.16012/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.16012","created_at":"2026-07-20T02:19:26.172437+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.16012v1","created_at":"2026-07-20T02:19:26.172437+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.16012","created_at":"2026-07-20T02:19:26.172437+00:00"},{"alias_kind":"pith_short_12","alias_value":"PARVOTX3UNYT","created_at":"2026-07-20T02:19:26.172437+00:00"},{"alias_kind":"pith_short_16","alias_value":"PARVOTX3UNYT3FS6","created_at":"2026-07-20T02:19:26.172437+00:00"},{"alias_kind":"pith_short_8","alias_value":"PARVOTX3","created_at":"2026-07-20T02:19:26.172437+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PARVOTX3UNYT3FS6OJIDCFXXII","json":"https://pith.science/pith/PARVOTX3UNYT3FS6OJIDCFXXII.json","graph_json":"https://pith.science/api/pith-number/PARVOTX3UNYT3FS6OJIDCFXXII/graph.json","events_json":"https://pith.science/api/pith-number/PARVOTX3UNYT3FS6OJIDCFXXII/events.json","paper":"https://pith.science/paper/PARVOTX3"},"agent_actions":{"view_html":"https://pith.science/pith/PARVOTX3UNYT3FS6OJIDCFXXII","download_json":"https://pith.science/pith/PARVOTX3UNYT3FS6OJIDCFXXII.json","view_paper":"https://pith.science/paper/PARVOTX3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.16012&json=true","fetch_graph":"https://pith.science/api/pith-number/PARVOTX3UNYT3FS6OJIDCFXXII/graph.json","fetch_events":"https://pith.science/api/pith-number/PARVOTX3UNYT3FS6OJIDCFXXII/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PARVOTX3UNYT3FS6OJIDCFXXII/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PARVOTX3UNYT3FS6OJIDCFXXII/action/storage_attestation","attest_author":"https://pith.science/pith/PARVOTX3UNYT3FS6OJIDCFXXII/action/author_attestation","sign_citation":"https://pith.science/pith/PARVOTX3UNYT3FS6OJIDCFXXII/action/citation_signature","submit_replication":"https://pith.science/pith/PARVOTX3UNYT3FS6OJIDCFXXII/action/replication_record"}},"created_at":"2026-07-20T02:19:26.172437+00:00","updated_at":"2026-07-20T02:19:26.172437+00:00"}