{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:35WBA2VZJOADPV4J7QBIQRPDBG","short_pith_number":"pith:35WBA2VZ","schema_version":"1.0","canonical_sha256":"df6c106ab94b8037d789fc028845e309b00a137339704a94122b428ffc72c0df","source":{"kind":"arxiv","id":"2506.09329","version":1},"attestation_state":"computed","paper":{"title":"Towards Efficient and Effective Alignment of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Yuxin Jiang","submitted_at":"2025-06-11T02:08:52Z","abstract_excerpt":"Large language models (LLMs) exhibit remarkable capabilities across diverse tasks, yet aligning them efficiently and effectively with human expectations remains a critical challenge. This thesis advances LLM alignment by introducing novel methodologies in data collection, training, and evaluation. We first address alignment data collection. Existing approaches rely heavily on manually curated datasets or proprietary models. To overcome these limitations, we propose Lion, an adversarial distillation framework that iteratively refines training data by identifying and generating challenging instr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09329","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-11T02:08:52Z","cross_cats_sorted":[],"title_canon_sha256":"ed6c2443c736ab3fa3ab7ffe6fdc6afc11f233b4d4871b80ca32fac70c060389","abstract_canon_sha256":"da75ea31328bb732d18b5d2800a95ad5f665cd73d2c046de43e85cb77e572965"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:34.899890Z","signature_b64":"Tt6xDfIIV20ISGMTA8ZyWdoNK2PZMHI6jzi7Wc+WDVbUt9rb875AEEequdaHYc0f+EIcrcZGSMEX1ygQ0xglCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"df6c106ab94b8037d789fc028845e309b00a137339704a94122b428ffc72c0df","last_reissued_at":"2026-07-05T11:19:34.899470Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:34.899470Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Efficient and Effective Alignment of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Yuxin Jiang","submitted_at":"2025-06-11T02:08:52Z","abstract_excerpt":"Large language models (LLMs) exhibit remarkable capabilities across diverse tasks, yet aligning them efficiently and effectively with human expectations remains a critical challenge. This thesis advances LLM alignment by introducing novel methodologies in data collection, training, and evaluation. We first address alignment data collection. Existing approaches rely heavily on manually curated datasets or proprietary models. To overcome these limitations, we propose Lion, an adversarial distillation framework that iteratively refines training data by identifying and generating challenging instr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09329","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09329/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09329","created_at":"2026-07-05T11:19:34.899520+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09329v1","created_at":"2026-07-05T11:19:34.899520+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09329","created_at":"2026-07-05T11:19:34.899520+00:00"},{"alias_kind":"pith_short_12","alias_value":"35WBA2VZJOAD","created_at":"2026-07-05T11:19:34.899520+00:00"},{"alias_kind":"pith_short_16","alias_value":"35WBA2VZJOADPV4J","created_at":"2026-07-05T11:19:34.899520+00:00"},{"alias_kind":"pith_short_8","alias_value":"35WBA2VZ","created_at":"2026-07-05T11:19:34.899520+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/35WBA2VZJOADPV4J7QBIQRPDBG","json":"https://pith.science/pith/35WBA2VZJOADPV4J7QBIQRPDBG.json","graph_json":"https://pith.science/api/pith-number/35WBA2VZJOADPV4J7QBIQRPDBG/graph.json","events_json":"https://pith.science/api/pith-number/35WBA2VZJOADPV4J7QBIQRPDBG/events.json","paper":"https://pith.science/paper/35WBA2VZ"},"agent_actions":{"view_html":"https://pith.science/pith/35WBA2VZJOADPV4J7QBIQRPDBG","download_json":"https://pith.science/pith/35WBA2VZJOADPV4J7QBIQRPDBG.json","view_paper":"https://pith.science/paper/35WBA2VZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09329&json=true","fetch_graph":"https://pith.science/api/pith-number/35WBA2VZJOADPV4J7QBIQRPDBG/graph.json","fetch_events":"https://pith.science/api/pith-number/35WBA2VZJOADPV4J7QBIQRPDBG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/35WBA2VZJOADPV4J7QBIQRPDBG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/35WBA2VZJOADPV4J7QBIQRPDBG/action/storage_attestation","attest_author":"https://pith.science/pith/35WBA2VZJOADPV4J7QBIQRPDBG/action/author_attestation","sign_citation":"https://pith.science/pith/35WBA2VZJOADPV4J7QBIQRPDBG/action/citation_signature","submit_replication":"https://pith.science/pith/35WBA2VZJOADPV4J7QBIQRPDBG/action/replication_record"}},"created_at":"2026-07-05T11:19:34.899520+00:00","updated_at":"2026-07-05T11:19:34.899520+00:00"}