{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XO34O6JHFBIOU2RNHKNWCICFTD","short_pith_number":"pith:XO34O6JH","schema_version":"1.0","canonical_sha256":"bbb7c779272850ea6a2d3a9b61204598f65129b336bd4ff3b66f6d7ee5efeb9b","source":{"kind":"arxiv","id":"2303.14408","version":1},"attestation_state":"computed","paper":{"title":"VL-SAT: Visual-Linguistic Semantics Assisted Training for 3D Semantic Scene Graph Prediction in Point Cloud","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowen Cheng, Dong Xu, Lichen Zhao, Lu Sheng, Yang Tang, Ziqin Wang","submitted_at":"2023-03-25T09:14:18Z","abstract_excerpt":"The task of 3D semantic scene graph (3DSSG) prediction in the point cloud is challenging since (1) the 3D point cloud only captures geometric structures with limited semantics compared to 2D images, and (2) long-tailed relation distribution inherently hinders the learning of unbiased prediction. Since 2D images provide rich semantics and scene graphs are in nature coped with languages, in this study, we propose Visual-Linguistic Semantics Assisted Training (VL-SAT) scheme that can significantly empower 3DSSG prediction models with discrimination about long-tailed and ambiguous semantic relatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.14408","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-03-25T09:14:18Z","cross_cats_sorted":[],"title_canon_sha256":"ef21779838c3b35e2c3d405abcdd196576b449a5b62de6f93423d6186aded2bd","abstract_canon_sha256":"3522f1febe1eac9b658acfb5e7c97be350879cdb0db642b0fb8d0f3192544497"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:54:41.158083Z","signature_b64":"rBPzOHUCqlWNp0OoRAWilLRt9AwjiBCmgODas/0IyJDSZCnn52ZRFCy/zrY1Oz8TvliQ7Hm9khcJceLbjytpDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bbb7c779272850ea6a2d3a9b61204598f65129b336bd4ff3b66f6d7ee5efeb9b","last_reissued_at":"2026-07-05T05:54:41.157742Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:54:41.157742Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VL-SAT: Visual-Linguistic Semantics Assisted Training for 3D Semantic Scene Graph Prediction in Point Cloud","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowen Cheng, Dong Xu, Lichen Zhao, Lu Sheng, Yang Tang, Ziqin Wang","submitted_at":"2023-03-25T09:14:18Z","abstract_excerpt":"The task of 3D semantic scene graph (3DSSG) prediction in the point cloud is challenging since (1) the 3D point cloud only captures geometric structures with limited semantics compared to 2D images, and (2) long-tailed relation distribution inherently hinders the learning of unbiased prediction. Since 2D images provide rich semantics and scene graphs are in nature coped with languages, in this study, we propose Visual-Linguistic Semantics Assisted Training (VL-SAT) scheme that can significantly empower 3DSSG prediction models with discrimination about long-tailed and ambiguous semantic relatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.14408","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.14408/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.14408","created_at":"2026-07-05T05:54:41.157799+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.14408v1","created_at":"2026-07-05T05:54:41.157799+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.14408","created_at":"2026-07-05T05:54:41.157799+00:00"},{"alias_kind":"pith_short_12","alias_value":"XO34O6JHFBIO","created_at":"2026-07-05T05:54:41.157799+00:00"},{"alias_kind":"pith_short_16","alias_value":"XO34O6JHFBIOU2RN","created_at":"2026-07-05T05:54:41.157799+00:00"},{"alias_kind":"pith_short_8","alias_value":"XO34O6JH","created_at":"2026-07-05T05:54:41.157799+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.21788","citing_title":"SceneGraphGrounder: Zero-Shot 3D Visual Grounding via Structured Scene Graph Matching","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XO34O6JHFBIOU2RNHKNWCICFTD","json":"https://pith.science/pith/XO34O6JHFBIOU2RNHKNWCICFTD.json","graph_json":"https://pith.science/api/pith-number/XO34O6JHFBIOU2RNHKNWCICFTD/graph.json","events_json":"https://pith.science/api/pith-number/XO34O6JHFBIOU2RNHKNWCICFTD/events.json","paper":"https://pith.science/paper/XO34O6JH"},"agent_actions":{"view_html":"https://pith.science/pith/XO34O6JHFBIOU2RNHKNWCICFTD","download_json":"https://pith.science/pith/XO34O6JHFBIOU2RNHKNWCICFTD.json","view_paper":"https://pith.science/paper/XO34O6JH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.14408&json=true","fetch_graph":"https://pith.science/api/pith-number/XO34O6JHFBIOU2RNHKNWCICFTD/graph.json","fetch_events":"https://pith.science/api/pith-number/XO34O6JHFBIOU2RNHKNWCICFTD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XO34O6JHFBIOU2RNHKNWCICFTD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XO34O6JHFBIOU2RNHKNWCICFTD/action/storage_attestation","attest_author":"https://pith.science/pith/XO34O6JHFBIOU2RNHKNWCICFTD/action/author_attestation","sign_citation":"https://pith.science/pith/XO34O6JHFBIOU2RNHKNWCICFTD/action/citation_signature","submit_replication":"https://pith.science/pith/XO34O6JHFBIOU2RNHKNWCICFTD/action/replication_record"}},"created_at":"2026-07-05T05:54:41.157799+00:00","updated_at":"2026-07-05T05:54:41.157799+00:00"}