{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL","short_pith_number":"pith:JI4LQ5AH","schema_version":"1.0","canonical_sha256":"4a38b87407574a9de5d3a4e41f97b972e74f28995381eb7869d677f616a1e84b","source":{"kind":"arxiv","id":"2409.08813","version":2},"attestation_state":"computed","paper":{"title":"Your Weak LLM is Secretly a Strong Teacher for Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Leitian Tao, Yixuan Li","submitted_at":"2024-09-13T13:24:52Z","abstract_excerpt":"The burgeoning capabilities of large language models (LLMs) have underscored the need for alignment to ensure these models act in accordance with human values and intentions. Existing alignment frameworks present constraints either in the form of expensive human effort or high computational costs. This paper explores a promising middle ground, where we employ a weak LLM that is significantly less resource-intensive than top-tier models, yet offers more automation than purely human feedback. We present a systematic study to evaluate and understand weak LLM's ability to generate feedback for ali"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.08813","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-13T13:24:52Z","cross_cats_sorted":[],"title_canon_sha256":"6a5a879736a68b641c5266b657d3fc33bf815ed9d8df58fef7071d08d725c17d","abstract_canon_sha256":"4b18ef6761e79070b318cff7fa6a840835815f0c7c046e7d0ab89cc72767ea4f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:47.319135Z","signature_b64":"/29dKpyfrDKYKUP2ohXG+EtzX4piOSo/YX+aCF9imxXh0ASOOenawk5SNLTsbpFo3ZEp9DrSBJkyvY5HfecCDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a38b87407574a9de5d3a4e41f97b972e74f28995381eb7869d677f616a1e84b","last_reissued_at":"2026-07-05T10:53:47.318538Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:47.318538Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Your Weak LLM is Secretly a Strong Teacher for Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Leitian Tao, Yixuan Li","submitted_at":"2024-09-13T13:24:52Z","abstract_excerpt":"The burgeoning capabilities of large language models (LLMs) have underscored the need for alignment to ensure these models act in accordance with human values and intentions. Existing alignment frameworks present constraints either in the form of expensive human effort or high computational costs. This paper explores a promising middle ground, where we employ a weak LLM that is significantly less resource-intensive than top-tier models, yet offers more automation than purely human feedback. We present a systematic study to evaluate and understand weak LLM's ability to generate feedback for ali"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.08813","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.08813/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.08813","created_at":"2026-07-05T10:53:47.318611+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.08813v2","created_at":"2026-07-05T10:53:47.318611+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.08813","created_at":"2026-07-05T10:53:47.318611+00:00"},{"alias_kind":"pith_short_12","alias_value":"JI4LQ5AHK5FJ","created_at":"2026-07-05T10:53:47.318611+00:00"},{"alias_kind":"pith_short_16","alias_value":"JI4LQ5AHK5FJ3ZOT","created_at":"2026-07-05T10:53:47.318611+00:00"},{"alias_kind":"pith_short_8","alias_value":"JI4LQ5AH","created_at":"2026-07-05T10:53:47.318611+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00424","citing_title":"Weak Critics Make Strong Learners: On-Policy Critique Distillation for Scalable Oversight","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":264,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":264,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05742","citing_title":"Weak-to-Strong Generalization is Nearly Inevitable (in Linear Models)","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL","json":"https://pith.science/pith/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL.json","graph_json":"https://pith.science/api/pith-number/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL/graph.json","events_json":"https://pith.science/api/pith-number/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL/events.json","paper":"https://pith.science/paper/JI4LQ5AH"},"agent_actions":{"view_html":"https://pith.science/pith/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL","download_json":"https://pith.science/pith/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL.json","view_paper":"https://pith.science/paper/JI4LQ5AH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.08813&json=true","fetch_graph":"https://pith.science/api/pith-number/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL/graph.json","fetch_events":"https://pith.science/api/pith-number/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL/action/storage_attestation","attest_author":"https://pith.science/pith/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL/action/author_attestation","sign_citation":"https://pith.science/pith/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL/action/citation_signature","submit_replication":"https://pith.science/pith/JI4LQ5AHK5FJ3ZOTUTSB7F5ZOL/action/replication_record"}},"created_at":"2026-07-05T10:53:47.318611+00:00","updated_at":"2026-07-05T10:53:47.318611+00:00"}