{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LBEYQ3RQPKEDSPYCGU2DMAYXRE","short_pith_number":"pith:LBEYQ3RQ","schema_version":"1.0","canonical_sha256":"5849886e307a88393f02353436031789182e712e5e8984c302c9da5b42c6f282","source":{"kind":"arxiv","id":"2312.01552","version":1},"attestation_state":"computed","paper":{"title":"The Unlocking Spell on Base LLMs: Rethinking Alignment via In-Context Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Abhilasha Ravichander, Bill Yuchen Lin, Chandra Bhagavatula, Khyathi Chandu, Melanie Sclar, Nouha Dziri, Ximing Lu, Yejin Choi","submitted_at":"2023-12-04T00:46:11Z","abstract_excerpt":"The alignment tuning process of large language models (LLMs) typically involves instruction learning through supervised fine-tuning (SFT) and preference tuning via reinforcement learning from human feedback (RLHF). A recent study, LIMA (Zhou et al. 2023), shows that using merely 1K examples for SFT can achieve significant alignment performance as well, suggesting that the effect of alignment tuning might be \"superficial.\" This raises questions about how exactly the alignment tuning transforms a base LLM.\n  We analyze the effect of alignment tuning by examining the token distribution shift betw"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.01552","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-12-04T00:46:11Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cf9cafa07ab6e084932bc5c93ef9c177e23aa3a92231bc5d86ddf7b45febd274","abstract_canon_sha256":"8c1e2f7e5a135f7a20e84a33c5b21dbb54bc33ac5b45f95a2767c13aa57f950c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:19:56.494249Z","signature_b64":"J9iZA/YK8v+ISz+ci603tCVUGXidycYSvgdIP6GYwRracO+V9Z75qRkxr5B7ZF+5mr8FX9kfXxXh9xZRNIHlBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5849886e307a88393f02353436031789182e712e5e8984c302c9da5b42c6f282","last_reissued_at":"2026-07-05T07:19:56.493832Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:19:56.493832Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Unlocking Spell on Base LLMs: Rethinking Alignment via In-Context Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Abhilasha Ravichander, Bill Yuchen Lin, Chandra Bhagavatula, Khyathi Chandu, Melanie Sclar, Nouha Dziri, Ximing Lu, Yejin Choi","submitted_at":"2023-12-04T00:46:11Z","abstract_excerpt":"The alignment tuning process of large language models (LLMs) typically involves instruction learning through supervised fine-tuning (SFT) and preference tuning via reinforcement learning from human feedback (RLHF). A recent study, LIMA (Zhou et al. 2023), shows that using merely 1K examples for SFT can achieve significant alignment performance as well, suggesting that the effect of alignment tuning might be \"superficial.\" This raises questions about how exactly the alignment tuning transforms a base LLM.\n  We analyze the effect of alignment tuning by examining the token distribution shift betw"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.01552","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.01552/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.01552","created_at":"2026-07-05T07:19:56.493887+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.01552v1","created_at":"2026-07-05T07:19:56.493887+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.01552","created_at":"2026-07-05T07:19:56.493887+00:00"},{"alias_kind":"pith_short_12","alias_value":"LBEYQ3RQPKED","created_at":"2026-07-05T07:19:56.493887+00:00"},{"alias_kind":"pith_short_16","alias_value":"LBEYQ3RQPKEDSPYC","created_at":"2026-07-05T07:19:56.493887+00:00"},{"alias_kind":"pith_short_8","alias_value":"LBEYQ3RQ","created_at":"2026-07-05T07:19:56.493887+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00531","citing_title":"Active-GRPO: Adaptive Imitation and Self-Improving Reasoning for Molecular Optimization","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07647","citing_title":"Quality-Conditioned Agreement in Automated Short Answer Scoring: Mid-Range Degradation and the Impact of Task-Specific Adaptation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07539","citing_title":"Prompt Governance? On Governing Technologies Governed by Natural Language","ref_index":206,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24154","citing_title":"Palette: A Modular, Controllable, and Efficient Framework for On-demand Authorized Safety Alignment Relaxation in LLMs","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29407","citing_title":"LC-ICL: Label-Guided Contrastive In-Context Learning for Robust Information Extraction","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29933","citing_title":"Towards Physical Intuitions for Alignment Dynamics: A Case Study With Randomness Crystallization","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23244","citing_title":"Convex Optimization for Alignment and Preference Learning on a Single GPU","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20129","citing_title":"SAID: Safety-Aware Intent Defense via Prefix Probing for Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2406.08464","citing_title":"Magpie: Alignment Data Synthesis from Scratch by Prompting Aligned LLMs with Nothing","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08405","citing_title":"Belief or Circuitry? Causal Evidence for In-Context Graph Learning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19018","citing_title":"Local Linearity of LLMs Enables Activation Steering via Model-Based Linear Optimal Control","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11050","citing_title":"Shared Emotion Geometry Across Small Language Models: A Cross-Architecture Study of Representation, Behavior, and Methodological Confounds","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LBEYQ3RQPKEDSPYCGU2DMAYXRE","json":"https://pith.science/pith/LBEYQ3RQPKEDSPYCGU2DMAYXRE.json","graph_json":"https://pith.science/api/pith-number/LBEYQ3RQPKEDSPYCGU2DMAYXRE/graph.json","events_json":"https://pith.science/api/pith-number/LBEYQ3RQPKEDSPYCGU2DMAYXRE/events.json","paper":"https://pith.science/paper/LBEYQ3RQ"},"agent_actions":{"view_html":"https://pith.science/pith/LBEYQ3RQPKEDSPYCGU2DMAYXRE","download_json":"https://pith.science/pith/LBEYQ3RQPKEDSPYCGU2DMAYXRE.json","view_paper":"https://pith.science/paper/LBEYQ3RQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.01552&json=true","fetch_graph":"https://pith.science/api/pith-number/LBEYQ3RQPKEDSPYCGU2DMAYXRE/graph.json","fetch_events":"https://pith.science/api/pith-number/LBEYQ3RQPKEDSPYCGU2DMAYXRE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LBEYQ3RQPKEDSPYCGU2DMAYXRE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LBEYQ3RQPKEDSPYCGU2DMAYXRE/action/storage_attestation","attest_author":"https://pith.science/pith/LBEYQ3RQPKEDSPYCGU2DMAYXRE/action/author_attestation","sign_citation":"https://pith.science/pith/LBEYQ3RQPKEDSPYCGU2DMAYXRE/action/citation_signature","submit_replication":"https://pith.science/pith/LBEYQ3RQPKEDSPYCGU2DMAYXRE/action/replication_record"}},"created_at":"2026-07-05T07:19:56.493887+00:00","updated_at":"2026-07-05T07:19:56.493887+00:00"}