{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:XHTMIH7MTOKDCHS7JGFES7UYCU","short_pith_number":"pith:XHTMIH7M","schema_version":"1.0","canonical_sha256":"b9e6c41fec9b94311e5f498a497e98151c6a3f64a933a7bfe543cb6acdb6e27a","source":{"kind":"arxiv","id":"2201.07366","version":2},"attestation_state":"computed","paper":{"title":"TriCoLo: Trimodal Contrastive Loss for Text to Shape Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Angel X. Chang, Han-Hung Lee, Ke Zhang, Yiming Zhang, Yue Ruan","submitted_at":"2022-01-19T00:15:15Z","abstract_excerpt":"Text-to-shape retrieval is an increasingly relevant problem with the growth of 3D shape data. Recent work on contrastive losses for learning joint embeddings over multimodal data has been successful at tasks such as retrieval and classification. Thus far, work on joint representation learning for 3D shapes and text has focused on improving embeddings through modeling of complex attention between representations, or multi-task learning. We propose a trimodal learning scheme over text, multi-view images and 3D shape voxels, and show that with large batch contrastive learning we achieve good perf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.07366","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-01-19T00:15:15Z","cross_cats_sorted":[],"title_canon_sha256":"9ce8e224bbe489d661bb3e4ffe9192ac7d0228ae71298620aad07f17195ce646","abstract_canon_sha256":"8f8f684777ab537b1c4d799fb86b147409d90aa2d15f0f2f68e3df35c72fc250"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:28:07.627357Z","signature_b64":"66cw3hIhVAH7+AttKHcwRc9BQ4HKBMo73moBWxbgqH5/U2EtT0tge7GnE7YA8LCO1+j2o6ZgYU/dIrKLF19UBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9e6c41fec9b94311e5f498a497e98151c6a3f64a933a7bfe543cb6acdb6e27a","last_reissued_at":"2026-07-05T07:28:07.626742Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:28:07.626742Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TriCoLo: Trimodal Contrastive Loss for Text to Shape Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Angel X. Chang, Han-Hung Lee, Ke Zhang, Yiming Zhang, Yue Ruan","submitted_at":"2022-01-19T00:15:15Z","abstract_excerpt":"Text-to-shape retrieval is an increasingly relevant problem with the growth of 3D shape data. Recent work on contrastive losses for learning joint embeddings over multimodal data has been successful at tasks such as retrieval and classification. Thus far, work on joint representation learning for 3D shapes and text has focused on improving embeddings through modeling of complex attention between representations, or multi-task learning. We propose a trimodal learning scheme over text, multi-view images and 3D shape voxels, and show that with large batch contrastive learning we achieve good perf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.07366","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.07366/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.07366","created_at":"2026-07-05T07:28:07.626805+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.07366v2","created_at":"2026-07-05T07:28:07.626805+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.07366","created_at":"2026-07-05T07:28:07.626805+00:00"},{"alias_kind":"pith_short_12","alias_value":"XHTMIH7MTOKD","created_at":"2026-07-05T07:28:07.626805+00:00"},{"alias_kind":"pith_short_16","alias_value":"XHTMIH7MTOKDCHS7","created_at":"2026-07-05T07:28:07.626805+00:00"},{"alias_kind":"pith_short_8","alias_value":"XHTMIH7M","created_at":"2026-07-05T07:28:07.626805+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.18460","citing_title":"Learning Invariant Modality Representation for Robust Multimodal Learning from a Causal Inference Perspective","ref_index":158,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XHTMIH7MTOKDCHS7JGFES7UYCU","json":"https://pith.science/pith/XHTMIH7MTOKDCHS7JGFES7UYCU.json","graph_json":"https://pith.science/api/pith-number/XHTMIH7MTOKDCHS7JGFES7UYCU/graph.json","events_json":"https://pith.science/api/pith-number/XHTMIH7MTOKDCHS7JGFES7UYCU/events.json","paper":"https://pith.science/paper/XHTMIH7M"},"agent_actions":{"view_html":"https://pith.science/pith/XHTMIH7MTOKDCHS7JGFES7UYCU","download_json":"https://pith.science/pith/XHTMIH7MTOKDCHS7JGFES7UYCU.json","view_paper":"https://pith.science/paper/XHTMIH7M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.07366&json=true","fetch_graph":"https://pith.science/api/pith-number/XHTMIH7MTOKDCHS7JGFES7UYCU/graph.json","fetch_events":"https://pith.science/api/pith-number/XHTMIH7MTOKDCHS7JGFES7UYCU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XHTMIH7MTOKDCHS7JGFES7UYCU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XHTMIH7MTOKDCHS7JGFES7UYCU/action/storage_attestation","attest_author":"https://pith.science/pith/XHTMIH7MTOKDCHS7JGFES7UYCU/action/author_attestation","sign_citation":"https://pith.science/pith/XHTMIH7MTOKDCHS7JGFES7UYCU/action/citation_signature","submit_replication":"https://pith.science/pith/XHTMIH7MTOKDCHS7JGFES7UYCU/action/replication_record"}},"created_at":"2026-07-05T07:28:07.626805+00:00","updated_at":"2026-07-05T07:28:07.626805+00:00"}