{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MTMPMRC6MZU4ZKJX4ZP4UGWVN4","short_pith_number":"pith:MTMPMRC6","schema_version":"1.0","canonical_sha256":"64d8f6445e6669cca937e65fca1ad56f2a6fdbda5ac0b546f6cc1047117071a9","source":{"kind":"arxiv","id":"2410.01957","version":2},"attestation_state":"computed","paper":{"title":"Challenges and Future Directions of Data-Centric AI Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jeffrey Wang, Leitian Tao, Min-Hsuan Yeh, Seongheon Park, Shawn Im, Xuefeng Du, Yixuan Li","submitted_at":"2024-10-02T19:03:42Z","abstract_excerpt":"As AI systems become increasingly capable and influential, ensuring their alignment with human values, preferences, and goals has become a critical research focus. Current alignment methods primarily focus on designing algorithms and loss functions but often underestimate the crucial role of data. This paper advocates for a shift towards data-centric AI alignment, emphasizing the need to enhance the quality and representativeness of data used in aligning AI systems. In this position paper, we highlight key challenges associated with both human-based and AI-based feedback within the data-centri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01957","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-02T19:03:42Z","cross_cats_sorted":[],"title_canon_sha256":"5c258a7cc4773ee9cdc41e329e2cc508493a19112fcb53fae27146599bb878fd","abstract_canon_sha256":"dd8ef08f883d079945492c69f4a84862a2f405f8ef0cabc9b232fa8f7d83f03a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:57:20.406679Z","signature_b64":"6FegRG+93s7v0sdbDr6mJJDvfEusEOGKYbnlB16p0Q52xS/JPqlu6tvnWgFdgLonsNfO6XbbzHuwyisUhjKoDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"64d8f6445e6669cca937e65fca1ad56f2a6fdbda5ac0b546f6cc1047117071a9","last_reissued_at":"2026-07-05T10:57:20.406187Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:57:20.406187Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Challenges and Future Directions of Data-Centric AI Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jeffrey Wang, Leitian Tao, Min-Hsuan Yeh, Seongheon Park, Shawn Im, Xuefeng Du, Yixuan Li","submitted_at":"2024-10-02T19:03:42Z","abstract_excerpt":"As AI systems become increasingly capable and influential, ensuring their alignment with human values, preferences, and goals has become a critical research focus. Current alignment methods primarily focus on designing algorithms and loss functions but often underestimate the crucial role of data. This paper advocates for a shift towards data-centric AI alignment, emphasizing the need to enhance the quality and representativeness of data used in aligning AI systems. In this position paper, we highlight key challenges associated with both human-based and AI-based feedback within the data-centri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01957","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01957/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01957","created_at":"2026-07-05T10:57:20.406245+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01957v2","created_at":"2026-07-05T10:57:20.406245+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01957","created_at":"2026-07-05T10:57:20.406245+00:00"},{"alias_kind":"pith_short_12","alias_value":"MTMPMRC6MZU4","created_at":"2026-07-05T10:57:20.406245+00:00"},{"alias_kind":"pith_short_16","alias_value":"MTMPMRC6MZU4ZKJX","created_at":"2026-07-05T10:57:20.406245+00:00"},{"alias_kind":"pith_short_8","alias_value":"MTMPMRC6","created_at":"2026-07-05T10:57:20.406245+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04136","citing_title":"A Technical Survey of Reinforcement Learning Techniques for Large Language Models","ref_index":20,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MTMPMRC6MZU4ZKJX4ZP4UGWVN4","json":"https://pith.science/pith/MTMPMRC6MZU4ZKJX4ZP4UGWVN4.json","graph_json":"https://pith.science/api/pith-number/MTMPMRC6MZU4ZKJX4ZP4UGWVN4/graph.json","events_json":"https://pith.science/api/pith-number/MTMPMRC6MZU4ZKJX4ZP4UGWVN4/events.json","paper":"https://pith.science/paper/MTMPMRC6"},"agent_actions":{"view_html":"https://pith.science/pith/MTMPMRC6MZU4ZKJX4ZP4UGWVN4","download_json":"https://pith.science/pith/MTMPMRC6MZU4ZKJX4ZP4UGWVN4.json","view_paper":"https://pith.science/paper/MTMPMRC6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01957&json=true","fetch_graph":"https://pith.science/api/pith-number/MTMPMRC6MZU4ZKJX4ZP4UGWVN4/graph.json","fetch_events":"https://pith.science/api/pith-number/MTMPMRC6MZU4ZKJX4ZP4UGWVN4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MTMPMRC6MZU4ZKJX4ZP4UGWVN4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MTMPMRC6MZU4ZKJX4ZP4UGWVN4/action/storage_attestation","attest_author":"https://pith.science/pith/MTMPMRC6MZU4ZKJX4ZP4UGWVN4/action/author_attestation","sign_citation":"https://pith.science/pith/MTMPMRC6MZU4ZKJX4ZP4UGWVN4/action/citation_signature","submit_replication":"https://pith.science/pith/MTMPMRC6MZU4ZKJX4ZP4UGWVN4/action/replication_record"}},"created_at":"2026-07-05T10:57:20.406245+00:00","updated_at":"2026-07-05T10:57:20.406245+00:00"}