{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:H2TBWP3S3TNOEP3COBPAAA62Y3","short_pith_number":"pith:H2TBWP3S","canonical_record":{"source":{"id":"2503.02832","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-04T17:57:09Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"65e6e4fed81a95f03ee6afa50771c05a24c5a43adbd6ba46b6fa55eb2972e14d","abstract_canon_sha256":"b09650687d1b20266ff0af0f2ff8fc3ead05985cd0f3e2058c2682c35e7e1817"},"schema_version":"1.0"},"canonical_sha256":"3ea61b3f72dcdae23f62705e0003dac6fa7ddc0df2d2f3134e9a96b3bd94aaa5","source":{"kind":"arxiv","id":"2503.02832","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2503.02832","created_at":"2026-07-05T11:41:39Z"},{"alias_kind":"arxiv_version","alias_value":"2503.02832v3","created_at":"2026-07-05T11:41:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.02832","created_at":"2026-07-05T11:41:39Z"},{"alias_kind":"pith_short_12","alias_value":"H2TBWP3S3TNO","created_at":"2026-07-05T11:41:39Z"},{"alias_kind":"pith_short_16","alias_value":"H2TBWP3S3TNOEP3C","created_at":"2026-07-05T11:41:39Z"},{"alias_kind":"pith_short_8","alias_value":"H2TBWP3S","created_at":"2026-07-05T11:41:39Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:H2TBWP3S3TNOEP3COBPAAA62Y3","target":"record","payload":{"canonical_record":{"source":{"id":"2503.02832","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-04T17:57:09Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"65e6e4fed81a95f03ee6afa50771c05a24c5a43adbd6ba46b6fa55eb2972e14d","abstract_canon_sha256":"b09650687d1b20266ff0af0f2ff8fc3ead05985cd0f3e2058c2682c35e7e1817"},"schema_version":"1.0"},"canonical_sha256":"3ea61b3f72dcdae23f62705e0003dac6fa7ddc0df2d2f3134e9a96b3bd94aaa5","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:41:39.846590Z","signature_b64":"asoStt0HRsmZG0RHCs71uz0HOhEmqBYPHGEMJwvenQcOps3MlnyHuWBKOFv8n7MKO+eaI+pjkE0/p58wvdODBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3ea61b3f72dcdae23f62705e0003dac6fa7ddc0df2d2f3134e9a96b3bd94aaa5","last_reissued_at":"2026-07-05T11:41:39.846103Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:41:39.846103Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2503.02832","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:41:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IWcJwUxsWPnE+x61O+ib94kc2DX9J7I4SHKkar0UV+6YsuX5rwO4YVUh79Ri5P1Z7yueRPyUZBt0COXrdWIpCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T06:25:43.964628Z"},"content_sha256":"0abb0282d4dd607c3db368aa60cfe08b5047c4cb50a3cdecc489a55adff295a1","schema_version":"1.0","event_id":"sha256:0abb0282d4dd607c3db368aa60cfe08b5047c4cb50a3cdecc489a55adff295a1"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:H2TBWP3S3TNOEP3COBPAAA62Y3","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"AlignDistil: Token-Level Language Model Alignment as Adaptive Policy Distillation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bojie Hu, Jinan Xu, Songming Zhang, Tong Zhang, Xue Zhang, Yufeng Chen","submitted_at":"2025-03-04T17:57:09Z","abstract_excerpt":"In modern large language models (LLMs), LLM alignment is of crucial importance and is typically achieved through methods such as reinforcement learning from human feedback (RLHF) and direct preference optimization (DPO). However, in most existing methods for LLM alignment, all tokens in the response are optimized using a sparse, response-level reward or preference annotation. The ignorance of token-level rewards may erroneously punish high-quality tokens or encourage low-quality tokens, resulting in suboptimal performance and slow convergence speed. To address this issue, we propose AlignDisti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.02832","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.02832/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:41:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"g4SSNsxbsordRhm59g7XDLsDAeCZxkMKPOeWMvhzwgSURwHmPOMXGCejU+TYelsI93DKrIoI8QNl7dtkCFzEDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T06:25:43.965582Z"},"content_sha256":"459cd4de8b9ca93cefb6f9e260a5f3727f6869f7dd766a1f1561bb57743096a1","schema_version":"1.0","event_id":"sha256:459cd4de8b9ca93cefb6f9e260a5f3727f6869f7dd766a1f1561bb57743096a1"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/H2TBWP3S3TNOEP3COBPAAA62Y3/bundle.json","state_url":"https://pith.science/pith/H2TBWP3S3TNOEP3COBPAAA62Y3/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/H2TBWP3S3TNOEP3COBPAAA62Y3/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T06:25:43Z","links":{"resolver":"https://pith.science/pith/H2TBWP3S3TNOEP3COBPAAA62Y3","bundle":"https://pith.science/pith/H2TBWP3S3TNOEP3COBPAAA62Y3/bundle.json","state":"https://pith.science/pith/H2TBWP3S3TNOEP3COBPAAA62Y3/state.json","well_known_bundle":"https://pith.science/.well-known/pith/H2TBWP3S3TNOEP3COBPAAA62Y3/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:H2TBWP3S3TNOEP3COBPAAA62Y3","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b09650687d1b20266ff0af0f2ff8fc3ead05985cd0f3e2058c2682c35e7e1817","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-04T17:57:09Z","title_canon_sha256":"65e6e4fed81a95f03ee6afa50771c05a24c5a43adbd6ba46b6fa55eb2972e14d"},"schema_version":"1.0","source":{"id":"2503.02832","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2503.02832","created_at":"2026-07-05T11:41:39Z"},{"alias_kind":"arxiv_version","alias_value":"2503.02832v3","created_at":"2026-07-05T11:41:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.02832","created_at":"2026-07-05T11:41:39Z"},{"alias_kind":"pith_short_12","alias_value":"H2TBWP3S3TNO","created_at":"2026-07-05T11:41:39Z"},{"alias_kind":"pith_short_16","alias_value":"H2TBWP3S3TNOEP3C","created_at":"2026-07-05T11:41:39Z"},{"alias_kind":"pith_short_8","alias_value":"H2TBWP3S","created_at":"2026-07-05T11:41:39Z"}],"graph_snapshots":[{"event_id":"sha256:459cd4de8b9ca93cefb6f9e260a5f3727f6869f7dd766a1f1561bb57743096a1","target":"graph","created_at":"2026-07-05T11:41:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2503.02832/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In modern large language models (LLMs), LLM alignment is of crucial importance and is typically achieved through methods such as reinforcement learning from human feedback (RLHF) and direct preference optimization (DPO). However, in most existing methods for LLM alignment, all tokens in the response are optimized using a sparse, response-level reward or preference annotation. The ignorance of token-level rewards may erroneously punish high-quality tokens or encourage low-quality tokens, resulting in suboptimal performance and slow convergence speed. To address this issue, we propose AlignDisti","authors_text":"Bojie Hu, Jinan Xu, Songming Zhang, Tong Zhang, Xue Zhang, Yufeng Chen","cross_cats":["cs.AI","cs.LG"],"headline":"","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-04T17:57:09Z","title":"AlignDistil: Token-Level Language Model Alignment as Adaptive Policy Distillation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.02832","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:0abb0282d4dd607c3db368aa60cfe08b5047c4cb50a3cdecc489a55adff295a1","target":"record","created_at":"2026-07-05T11:41:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b09650687d1b20266ff0af0f2ff8fc3ead05985cd0f3e2058c2682c35e7e1817","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-04T17:57:09Z","title_canon_sha256":"65e6e4fed81a95f03ee6afa50771c05a24c5a43adbd6ba46b6fa55eb2972e14d"},"schema_version":"1.0","source":{"id":"2503.02832","kind":"arxiv","version":3}},"canonical_sha256":"3ea61b3f72dcdae23f62705e0003dac6fa7ddc0df2d2f3134e9a96b3bd94aaa5","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3ea61b3f72dcdae23f62705e0003dac6fa7ddc0df2d2f3134e9a96b3bd94aaa5","first_computed_at":"2026-07-05T11:41:39.846103Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:41:39.846103Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"asoStt0HRsmZG0RHCs71uz0HOhEmqBYPHGEMJwvenQcOps3MlnyHuWBKOFv8n7MKO+eaI+pjkE0/p58wvdODBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:41:39.846590Z","signed_message":"canonical_sha256_bytes"},"source_id":"2503.02832","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:0abb0282d4dd607c3db368aa60cfe08b5047c4cb50a3cdecc489a55adff295a1","sha256:459cd4de8b9ca93cefb6f9e260a5f3727f6869f7dd766a1f1561bb57743096a1"],"state_sha256":"4cd45443c25132e56671bbb32f3b5671a6053ba8eb645c552b10447c011fc607"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"kKdgF3c3M8MyR7He3KfCwOX38Lum2MpGOV7+ov50UWjPv5vJMVGWxkB0ywA08KNrqXMmXwxAXVDUVPuiIDS/CQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T06:25:43.971343Z","bundle_sha256":"185528c70e34c1180851ac96d4462ea493b88291d7742f2d2d3bb166cb8d2e3a"}}