{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:URCTKSSNMY5R5SKB5QOAGIF3XU","short_pith_number":"pith:URCTKSSN","schema_version":"1.0","canonical_sha256":"a445354a4d663b1ec941ec1c0320bbbd0efce51261b18e1112e72cbbc03ae6b9","source":{"kind":"arxiv","id":"2506.09378","version":1},"attestation_state":"computed","paper":{"title":"UniForward: Unified 3D Scene and Semantic Field Reconstruction via Feed-Forward Gaussian Splatting from Only Sparse-View Images","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jingyu Gong, Lizhuang Ma, Qijian Tian, Xin Tan, Yuan Xie","submitted_at":"2025-06-11T04:01:21Z","abstract_excerpt":"We propose a feed-forward Gaussian Splatting model that unifies 3D scene and semantic field reconstruction. Combining 3D scenes with semantic fields facilitates the perception and understanding of the surrounding environment. However, key challenges include embedding semantics into 3D representations, achieving generalizable real-time reconstruction, and ensuring practical applicability by using only images as input without camera parameters or ground truth depth. To this end, we propose UniForward, a feed-forward model to predict 3D Gaussians with anisotropic semantic features from only uncal"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09378","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-11T04:01:21Z","cross_cats_sorted":[],"title_canon_sha256":"6d2bcf9bc29dfaf07c7e5618ed9209ef883cc6d077d8765439d12cbd72ff3f42","abstract_canon_sha256":"e99729c3633b91ca0cadab6ff05ce5e181fa3ad563ab0f46d0c0220c15d67f6b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:36.098227Z","signature_b64":"3bW7Ny2VSpZQb9anYCNC+5+L8nJbQhE37tYlX7umEcyJJdryu+Tej1jpaRzTcwO4qLcJCLhwSZuhUQfvTHxTDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a445354a4d663b1ec941ec1c0320bbbd0efce51261b18e1112e72cbbc03ae6b9","last_reissued_at":"2026-07-05T11:19:36.097501Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:36.097501Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"UniForward: Unified 3D Scene and Semantic Field Reconstruction via Feed-Forward Gaussian Splatting from Only Sparse-View Images","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jingyu Gong, Lizhuang Ma, Qijian Tian, Xin Tan, Yuan Xie","submitted_at":"2025-06-11T04:01:21Z","abstract_excerpt":"We propose a feed-forward Gaussian Splatting model that unifies 3D scene and semantic field reconstruction. Combining 3D scenes with semantic fields facilitates the perception and understanding of the surrounding environment. However, key challenges include embedding semantics into 3D representations, achieving generalizable real-time reconstruction, and ensuring practical applicability by using only images as input without camera parameters or ground truth depth. To this end, we propose UniForward, a feed-forward model to predict 3D Gaussians with anisotropic semantic features from only uncal"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09378","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09378/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09378","created_at":"2026-07-05T11:19:36.097576+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09378v1","created_at":"2026-07-05T11:19:36.097576+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09378","created_at":"2026-07-05T11:19:36.097576+00:00"},{"alias_kind":"pith_short_12","alias_value":"URCTKSSNMY5R","created_at":"2026-07-05T11:19:36.097576+00:00"},{"alias_kind":"pith_short_16","alias_value":"URCTKSSNMY5R5SKB","created_at":"2026-07-05T11:19:36.097576+00:00"},{"alias_kind":"pith_short_8","alias_value":"URCTKSSN","created_at":"2026-07-05T11:19:36.097576+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01633","citing_title":"Bridging 3D Gaussians and Semantic Occupancy for Comprehensive Open-Vocabulary Scene Understanding from Unposed Images","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2512.17541","citing_title":"FLEG: Feed-Forward Language Embedded Gaussian Splatting from Any Views via Compact Semantic Representation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10573","citing_title":"Learning 3D Representations for Spatial Intelligence from Unposed Multi-View Images","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09862","citing_title":"FF3R: Feedforward Feature 3D Reconstruction from Unconstrained views","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/URCTKSSNMY5R5SKB5QOAGIF3XU","json":"https://pith.science/pith/URCTKSSNMY5R5SKB5QOAGIF3XU.json","graph_json":"https://pith.science/api/pith-number/URCTKSSNMY5R5SKB5QOAGIF3XU/graph.json","events_json":"https://pith.science/api/pith-number/URCTKSSNMY5R5SKB5QOAGIF3XU/events.json","paper":"https://pith.science/paper/URCTKSSN"},"agent_actions":{"view_html":"https://pith.science/pith/URCTKSSNMY5R5SKB5QOAGIF3XU","download_json":"https://pith.science/pith/URCTKSSNMY5R5SKB5QOAGIF3XU.json","view_paper":"https://pith.science/paper/URCTKSSN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09378&json=true","fetch_graph":"https://pith.science/api/pith-number/URCTKSSNMY5R5SKB5QOAGIF3XU/graph.json","fetch_events":"https://pith.science/api/pith-number/URCTKSSNMY5R5SKB5QOAGIF3XU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/URCTKSSNMY5R5SKB5QOAGIF3XU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/URCTKSSNMY5R5SKB5QOAGIF3XU/action/storage_attestation","attest_author":"https://pith.science/pith/URCTKSSNMY5R5SKB5QOAGIF3XU/action/author_attestation","sign_citation":"https://pith.science/pith/URCTKSSNMY5R5SKB5QOAGIF3XU/action/citation_signature","submit_replication":"https://pith.science/pith/URCTKSSNMY5R5SKB5QOAGIF3XU/action/replication_record"}},"created_at":"2026-07-05T11:19:36.097576+00:00","updated_at":"2026-07-05T11:19:36.097576+00:00"}