{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BIZQAIX4ZMNOTTGLVCMXU2CYQT","short_pith_number":"pith:BIZQAIX4","schema_version":"1.0","canonical_sha256":"0a330022fccb1ae9cccba8997a685884e19b51b19e70b37ad4520e3dbfbf1966","source":{"kind":"arxiv","id":"2410.12621","version":2},"attestation_state":"computed","paper":{"title":"Weak-to-Strong Generalization beyond Accuracy: a Pilot Study in Safety, Toxicity, and Legal Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Bo Hui, Ruimeng Ye, Yang Xiao","submitted_at":"2024-10-16T14:40:32Z","abstract_excerpt":"As large language models (LLMs) continue to advance, ensuring their alignment with human values becomes increasingly critical. Traditional alignment methods heavily rely on human feedback to fine-tune models. With the emergence of superhuman models whose outputs may surpass human understanding, evaluating and aligning these models using human judgments poses significant challenges. To address the challenges, recent works use weak supervisors to elicit knowledge from much stronger models. However, there are important disanalogies between the empirical setup in the existing works and the genuine"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12621","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-16T14:40:32Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"7574033f519edb86f80bd377f4eddc692a9f4403c8a5bbbfb2cc964f2e3fab13","abstract_canon_sha256":"7271e4e7cc7ccbbd43462bd91a3998db041683faec48154f4e0aa46f128c301f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:36.909576Z","signature_b64":"0IiWsHgH9yQyTjqiCsMiy0dzwebj4WHKjk1DSSVkZPfAox7FM4EHc8VV0IMAGHN+apMB8/IiEyojTK+OhCoPAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0a330022fccb1ae9cccba8997a685884e19b51b19e70b37ad4520e3dbfbf1966","last_reissued_at":"2026-07-05T10:38:36.909071Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:36.909071Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Weak-to-Strong Generalization beyond Accuracy: a Pilot Study in Safety, Toxicity, and Legal Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Bo Hui, Ruimeng Ye, Yang Xiao","submitted_at":"2024-10-16T14:40:32Z","abstract_excerpt":"As large language models (LLMs) continue to advance, ensuring their alignment with human values becomes increasingly critical. Traditional alignment methods heavily rely on human feedback to fine-tune models. With the emergence of superhuman models whose outputs may surpass human understanding, evaluating and aligning these models using human judgments poses significant challenges. To address the challenges, recent works use weak supervisors to elicit knowledge from much stronger models. However, there are important disanalogies between the empirical setup in the existing works and the genuine"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12621","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12621/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12621","created_at":"2026-07-05T10:38:36.909131+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12621v2","created_at":"2026-07-05T10:38:36.909131+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12621","created_at":"2026-07-05T10:38:36.909131+00:00"},{"alias_kind":"pith_short_12","alias_value":"BIZQAIX4ZMNO","created_at":"2026-07-05T10:38:36.909131+00:00"},{"alias_kind":"pith_short_16","alias_value":"BIZQAIX4ZMNOTTGL","created_at":"2026-07-05T10:38:36.909131+00:00"},{"alias_kind":"pith_short_8","alias_value":"BIZQAIX4","created_at":"2026-07-05T10:38:36.909131+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.05710","citing_title":"On the Blessing of Pre-training in Weak-to-Strong Generalization","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BIZQAIX4ZMNOTTGLVCMXU2CYQT","json":"https://pith.science/pith/BIZQAIX4ZMNOTTGLVCMXU2CYQT.json","graph_json":"https://pith.science/api/pith-number/BIZQAIX4ZMNOTTGLVCMXU2CYQT/graph.json","events_json":"https://pith.science/api/pith-number/BIZQAIX4ZMNOTTGLVCMXU2CYQT/events.json","paper":"https://pith.science/paper/BIZQAIX4"},"agent_actions":{"view_html":"https://pith.science/pith/BIZQAIX4ZMNOTTGLVCMXU2CYQT","download_json":"https://pith.science/pith/BIZQAIX4ZMNOTTGLVCMXU2CYQT.json","view_paper":"https://pith.science/paper/BIZQAIX4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12621&json=true","fetch_graph":"https://pith.science/api/pith-number/BIZQAIX4ZMNOTTGLVCMXU2CYQT/graph.json","fetch_events":"https://pith.science/api/pith-number/BIZQAIX4ZMNOTTGLVCMXU2CYQT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BIZQAIX4ZMNOTTGLVCMXU2CYQT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BIZQAIX4ZMNOTTGLVCMXU2CYQT/action/storage_attestation","attest_author":"https://pith.science/pith/BIZQAIX4ZMNOTTGLVCMXU2CYQT/action/author_attestation","sign_citation":"https://pith.science/pith/BIZQAIX4ZMNOTTGLVCMXU2CYQT/action/citation_signature","submit_replication":"https://pith.science/pith/BIZQAIX4ZMNOTTGLVCMXU2CYQT/action/replication_record"}},"created_at":"2026-07-05T10:38:36.909131+00:00","updated_at":"2026-07-05T10:38:36.909131+00:00"}