{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:326EREQXZ4HUOY3AIPR5XXC46Z","short_pith_number":"pith:326EREQX","schema_version":"1.0","canonical_sha256":"debc489217cf0f47636043e3dbdc5cf6759d9d3080568376f96b05b7d930792f","source":{"kind":"arxiv","id":"2501.00418","version":1},"attestation_state":"computed","paper":{"title":"Generalizing Trust: Weak-to-Strong Trustworthiness in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aounon Kumar, Himabindu Lakkaraju, Lillian Sun, Martin Pawelczyk, Zhenting Qi","submitted_at":"2024-12-31T12:40:02Z","abstract_excerpt":"The rapid proliferation of generative AI, especially large language models, has led to their integration into a variety of applications. A key phenomenon known as weak-to-strong generalization - where a strong model trained on a weak model's outputs surpasses the weak model in task performance - has gained significant attention. Yet, whether critical trustworthiness properties such as robustness, fairness, and privacy can generalize similarly remains an open question. In this work, we study this question by examining if a stronger model can inherit trustworthiness properties when fine-tuned on"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.00418","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-31T12:40:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"346bd1be5d4fac5a07399c296fb43a4f3ad4d8994798b7784dc54c4d8e8fc4a3","abstract_canon_sha256":"552f7437746b14423efffe1933151b71126368992e381952397b0016636532f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:56:05.623528Z","signature_b64":"4wTuY6esMn9JLzQksz/rpqrmG0+m3RCnZHZ8X9AN4mxuTfzq92jv4k1IxGG98dMmoFu+CtV3cvEKoyCKvEb8DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"debc489217cf0f47636043e3dbdc5cf6759d9d3080568376f96b05b7d930792f","last_reissued_at":"2026-07-05T09:56:05.622956Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:56:05.622956Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalizing Trust: Weak-to-Strong Trustworthiness in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aounon Kumar, Himabindu Lakkaraju, Lillian Sun, Martin Pawelczyk, Zhenting Qi","submitted_at":"2024-12-31T12:40:02Z","abstract_excerpt":"The rapid proliferation of generative AI, especially large language models, has led to their integration into a variety of applications. A key phenomenon known as weak-to-strong generalization - where a strong model trained on a weak model's outputs surpasses the weak model in task performance - has gained significant attention. Yet, whether critical trustworthiness properties such as robustness, fairness, and privacy can generalize similarly remains an open question. In this work, we study this question by examining if a stronger model can inherit trustworthiness properties when fine-tuned on"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.00418","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.00418/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.00418","created_at":"2026-07-05T09:56:05.623025+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.00418v1","created_at":"2026-07-05T09:56:05.623025+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.00418","created_at":"2026-07-05T09:56:05.623025+00:00"},{"alias_kind":"pith_short_12","alias_value":"326EREQXZ4HU","created_at":"2026-07-05T09:56:05.623025+00:00"},{"alias_kind":"pith_short_16","alias_value":"326EREQXZ4HUOY3A","created_at":"2026-07-05T09:56:05.623025+00:00"},{"alias_kind":"pith_short_8","alias_value":"326EREQX","created_at":"2026-07-05T09:56:05.623025+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01000","citing_title":"Trust Functions: Near-Lossless Weak-to-Strong Generalization by Learning When to Trust the Weak Teacher","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05710","citing_title":"On the Blessing of Pre-training in Weak-to-Strong Generalization","ref_index":101,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/326EREQXZ4HUOY3AIPR5XXC46Z","json":"https://pith.science/pith/326EREQXZ4HUOY3AIPR5XXC46Z.json","graph_json":"https://pith.science/api/pith-number/326EREQXZ4HUOY3AIPR5XXC46Z/graph.json","events_json":"https://pith.science/api/pith-number/326EREQXZ4HUOY3AIPR5XXC46Z/events.json","paper":"https://pith.science/paper/326EREQX"},"agent_actions":{"view_html":"https://pith.science/pith/326EREQXZ4HUOY3AIPR5XXC46Z","download_json":"https://pith.science/pith/326EREQXZ4HUOY3AIPR5XXC46Z.json","view_paper":"https://pith.science/paper/326EREQX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.00418&json=true","fetch_graph":"https://pith.science/api/pith-number/326EREQXZ4HUOY3AIPR5XXC46Z/graph.json","fetch_events":"https://pith.science/api/pith-number/326EREQXZ4HUOY3AIPR5XXC46Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/326EREQXZ4HUOY3AIPR5XXC46Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/326EREQXZ4HUOY3AIPR5XXC46Z/action/storage_attestation","attest_author":"https://pith.science/pith/326EREQXZ4HUOY3AIPR5XXC46Z/action/author_attestation","sign_citation":"https://pith.science/pith/326EREQXZ4HUOY3AIPR5XXC46Z/action/citation_signature","submit_replication":"https://pith.science/pith/326EREQXZ4HUOY3AIPR5XXC46Z/action/replication_record"}},"created_at":"2026-07-05T09:56:05.623025+00:00","updated_at":"2026-07-05T09:56:05.623025+00:00"}