{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2OGR34YYKZRTOBYWGB7XF4F2NS","short_pith_number":"pith:2OGR34YY","schema_version":"1.0","canonical_sha256":"d38d1df3185663370716307f72f0ba6cb53582b8371b2cfdb0e2ccb6ca0d79f6","source":{"kind":"arxiv","id":"2502.17927","version":2},"attestation_state":"computed","paper":{"title":"Advantage-Guided Distillation for Preference Alignment in Small Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fanqi Wan, Jiajian Guo, Qifan Wang, Shiping Gao, Xiaojun Quan","submitted_at":"2025-02-25T07:47:22Z","abstract_excerpt":"Alignment techniques enable Large Language Models (LLMs) to generate outputs that align with human preferences and play a crucial role in their effectiveness. However, their impact often diminishes when applied to Small Language Models (SLMs), likely due to the limited capacity of these models. Instead of directly applying existing alignment techniques to SLMs, we propose to utilize a well-aligned teacher LLM to guide the alignment process for these models, thereby facilitating the transfer of the teacher's knowledge of human preferences to the student model. To achieve this, we first explore "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.17927","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-25T07:47:22Z","cross_cats_sorted":[],"title_canon_sha256":"0450b766f74206029f06dafddeb54ecd7aa2a249b749cd0e6073aba4289e9fdb","abstract_canon_sha256":"2049ca2d814d687904faf12b2086b95800ff5958efb2c957c5db5927d01fc834"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:24:31.884266Z","signature_b64":"5NhYj+GrSeFmZuBHpoxo/2cCiPOiXBNSWZrHmXfglRCyr3t9Tqf8aRJO2FmS8qIKyrzAxY/atrdUnhF+6lZgCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d38d1df3185663370716307f72f0ba6cb53582b8371b2cfdb0e2ccb6ca0d79f6","last_reissued_at":"2026-07-05T10:24:31.883525Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:24:31.883525Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Advantage-Guided Distillation for Preference Alignment in Small Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fanqi Wan, Jiajian Guo, Qifan Wang, Shiping Gao, Xiaojun Quan","submitted_at":"2025-02-25T07:47:22Z","abstract_excerpt":"Alignment techniques enable Large Language Models (LLMs) to generate outputs that align with human preferences and play a crucial role in their effectiveness. However, their impact often diminishes when applied to Small Language Models (SLMs), likely due to the limited capacity of these models. Instead of directly applying existing alignment techniques to SLMs, we propose to utilize a well-aligned teacher LLM to guide the alignment process for these models, thereby facilitating the transfer of the teacher's knowledge of human preferences to the student model. To achieve this, we first explore "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.17927","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.17927/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.17927","created_at":"2026-07-05T10:24:31.883580+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.17927v2","created_at":"2026-07-05T10:24:31.883580+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.17927","created_at":"2026-07-05T10:24:31.883580+00:00"},{"alias_kind":"pith_short_12","alias_value":"2OGR34YYKZRT","created_at":"2026-07-05T10:24:31.883580+00:00"},{"alias_kind":"pith_short_16","alias_value":"2OGR34YYKZRTOBYW","created_at":"2026-07-05T10:24:31.883580+00:00"},{"alias_kind":"pith_short_8","alias_value":"2OGR34YY","created_at":"2026-07-05T10:24:31.883580+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.00054","citing_title":"Enhancing Reasoning Capabilities in SLMs with Reward Guided Dataset Distillation","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2OGR34YYKZRTOBYWGB7XF4F2NS","json":"https://pith.science/pith/2OGR34YYKZRTOBYWGB7XF4F2NS.json","graph_json":"https://pith.science/api/pith-number/2OGR34YYKZRTOBYWGB7XF4F2NS/graph.json","events_json":"https://pith.science/api/pith-number/2OGR34YYKZRTOBYWGB7XF4F2NS/events.json","paper":"https://pith.science/paper/2OGR34YY"},"agent_actions":{"view_html":"https://pith.science/pith/2OGR34YYKZRTOBYWGB7XF4F2NS","download_json":"https://pith.science/pith/2OGR34YYKZRTOBYWGB7XF4F2NS.json","view_paper":"https://pith.science/paper/2OGR34YY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.17927&json=true","fetch_graph":"https://pith.science/api/pith-number/2OGR34YYKZRTOBYWGB7XF4F2NS/graph.json","fetch_events":"https://pith.science/api/pith-number/2OGR34YYKZRTOBYWGB7XF4F2NS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2OGR34YYKZRTOBYWGB7XF4F2NS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2OGR34YYKZRTOBYWGB7XF4F2NS/action/storage_attestation","attest_author":"https://pith.science/pith/2OGR34YYKZRTOBYWGB7XF4F2NS/action/author_attestation","sign_citation":"https://pith.science/pith/2OGR34YYKZRTOBYWGB7XF4F2NS/action/citation_signature","submit_replication":"https://pith.science/pith/2OGR34YYKZRTOBYWGB7XF4F2NS/action/replication_record"}},"created_at":"2026-07-05T10:24:31.883580+00:00","updated_at":"2026-07-05T10:24:31.883580+00:00"}