{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3B5KEBEHMXHFCWGTYM35UUPIEG","short_pith_number":"pith:3B5KEBEH","schema_version":"1.0","canonical_sha256":"d87aa2048765ce5158d3c337da51e821ac5e16e19ca0b309765bd3121dadc055","source":{"kind":"arxiv","id":"2401.16184","version":6},"attestation_state":"computed","paper":{"title":"Vocabulary-Defined Semantics: Latent Space Clustering for Improving In-Context Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aldeida Aleti, Chunyang Chen, Hongyu Zhang, Jian Gu","submitted_at":"2024-01-29T14:29:48Z","abstract_excerpt":"In-context learning enables language models (LM) to adapt to downstream data or tasks by incorporating few samples as demonstrations within the prompts. It offers strong performance without the expense of fine-tuning. However, the performance of in-context learning can be unstable depending on the quality, format, or order of demonstrations, which in turn exacerbates the difficulty of optimization. Prior work, such as Knn Prompting, index samples based on the similarities of logits at the output-side, in addition to the regular retrieval operation at the input-side. They improve in-context lea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.16184","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-29T14:29:48Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"f72a975966486d1f209ef4dd5c8be57fa47a6b90a69623a7cc436152e114ffdd","abstract_canon_sha256":"43bb92f6d3631cea7e35ea1306228c63e5b21d2df465171ee7cf4f543fbececf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:53.792266Z","signature_b64":"jq5fDpyWDklZPfCeGKlhT7Tgj4+FD9mUiLrpWiAtua9hYrITwWWNo65Ebt2zW/HffL0OpaNkPkGMt5d+ya5hCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d87aa2048765ce5158d3c337da51e821ac5e16e19ca0b309765bd3121dadc055","last_reissued_at":"2026-07-05T09:19:53.791821Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:53.791821Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vocabulary-Defined Semantics: Latent Space Clustering for Improving In-Context Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aldeida Aleti, Chunyang Chen, Hongyu Zhang, Jian Gu","submitted_at":"2024-01-29T14:29:48Z","abstract_excerpt":"In-context learning enables language models (LM) to adapt to downstream data or tasks by incorporating few samples as demonstrations within the prompts. It offers strong performance without the expense of fine-tuning. However, the performance of in-context learning can be unstable depending on the quality, format, or order of demonstrations, which in turn exacerbates the difficulty of optimization. Prior work, such as Knn Prompting, index samples based on the similarities of logits at the output-side, in addition to the regular retrieval operation at the input-side. They improve in-context lea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.16184","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.16184/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.16184","created_at":"2026-07-05T09:19:53.791874+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.16184v6","created_at":"2026-07-05T09:19:53.791874+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.16184","created_at":"2026-07-05T09:19:53.791874+00:00"},{"alias_kind":"pith_short_12","alias_value":"3B5KEBEHMXHF","created_at":"2026-07-05T09:19:53.791874+00:00"},{"alias_kind":"pith_short_16","alias_value":"3B5KEBEHMXHFCWGT","created_at":"2026-07-05T09:19:53.791874+00:00"},{"alias_kind":"pith_short_8","alias_value":"3B5KEBEH","created_at":"2026-07-05T09:19:53.791874+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.32022","citing_title":"SemRF: A Semantic Reference Frame for Residual-Stream Dynamics in Language Models","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3B5KEBEHMXHFCWGTYM35UUPIEG","json":"https://pith.science/pith/3B5KEBEHMXHFCWGTYM35UUPIEG.json","graph_json":"https://pith.science/api/pith-number/3B5KEBEHMXHFCWGTYM35UUPIEG/graph.json","events_json":"https://pith.science/api/pith-number/3B5KEBEHMXHFCWGTYM35UUPIEG/events.json","paper":"https://pith.science/paper/3B5KEBEH"},"agent_actions":{"view_html":"https://pith.science/pith/3B5KEBEHMXHFCWGTYM35UUPIEG","download_json":"https://pith.science/pith/3B5KEBEHMXHFCWGTYM35UUPIEG.json","view_paper":"https://pith.science/paper/3B5KEBEH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.16184&json=true","fetch_graph":"https://pith.science/api/pith-number/3B5KEBEHMXHFCWGTYM35UUPIEG/graph.json","fetch_events":"https://pith.science/api/pith-number/3B5KEBEHMXHFCWGTYM35UUPIEG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3B5KEBEHMXHFCWGTYM35UUPIEG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3B5KEBEHMXHFCWGTYM35UUPIEG/action/storage_attestation","attest_author":"https://pith.science/pith/3B5KEBEHMXHFCWGTYM35UUPIEG/action/author_attestation","sign_citation":"https://pith.science/pith/3B5KEBEHMXHFCWGTYM35UUPIEG/action/citation_signature","submit_replication":"https://pith.science/pith/3B5KEBEHMXHFCWGTYM35UUPIEG/action/replication_record"}},"created_at":"2026-07-05T09:19:53.791874+00:00","updated_at":"2026-07-05T09:19:53.791874+00:00"}