{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:4ON4L6KDUPQKSYICFQSM4YREMC","short_pith_number":"pith:4ON4L6KD","schema_version":"1.0","canonical_sha256":"e39bc5f943a3e0a961022c24ce622460b2b6adad0d044dd093c4816e9aa21af5","source":{"kind":"arxiv","id":"2308.05696","version":2},"attestation_state":"computed","paper":{"title":"A Preliminary Study of the Intrinsic Relationship between Complexity and Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Binyuan Hui, Bowen Yu, Fei Huang, Haiyang Yu, Nevin L. Zhang, Yingxiu Zhao, Yongbin Li","submitted_at":"2023-08-10T16:58:51Z","abstract_excerpt":"Training large language models (LLMs) with open-domain instruction data has yielded remarkable success in aligning to end tasks and human preferences. Extensive research has highlighted the importance of the quality and diversity of instruction data. However, the impact of data complexity, as a crucial metric, remains relatively unexplored from three aspects: (1)where the sustainability of performance improvements with increasing complexity is uncertain; (2)whether the improvement brought by complexity merely comes from introducing more training tokens; and (3)where the potential benefits of i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.05696","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-08-10T16:58:51Z","cross_cats_sorted":[],"title_canon_sha256":"9f3b9904e3fc730114dcbf4a5ac58fa03506c245fa95d1a8f8e833d98b2055fb","abstract_canon_sha256":"5df02b5da4f5e2a30d1a2f18c789e9d1c53e6965dbc69f28cbe3a66fb3a96103"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:50:20.001165Z","signature_b64":"lDtpyOsASIKGZmkcqV0IsL+/f/NQlwDxsQi+1vnnLnnpKxmz+buKCPQm1gwN3o7uwbdcTu7fZZitAWc0U9WWBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e39bc5f943a3e0a961022c24ce622460b2b6adad0d044dd093c4816e9aa21af5","last_reissued_at":"2026-07-05T07:50:20.000747Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:50:20.000747Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Preliminary Study of the Intrinsic Relationship between Complexity and Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Binyuan Hui, Bowen Yu, Fei Huang, Haiyang Yu, Nevin L. Zhang, Yingxiu Zhao, Yongbin Li","submitted_at":"2023-08-10T16:58:51Z","abstract_excerpt":"Training large language models (LLMs) with open-domain instruction data has yielded remarkable success in aligning to end tasks and human preferences. Extensive research has highlighted the importance of the quality and diversity of instruction data. However, the impact of data complexity, as a crucial metric, remains relatively unexplored from three aspects: (1)where the sustainability of performance improvements with increasing complexity is uncertain; (2)whether the improvement brought by complexity merely comes from introducing more training tokens; and (3)where the potential benefits of i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.05696","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.05696/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.05696","created_at":"2026-07-05T07:50:20.000805+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.05696v2","created_at":"2026-07-05T07:50:20.000805+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.05696","created_at":"2026-07-05T07:50:20.000805+00:00"},{"alias_kind":"pith_short_12","alias_value":"4ON4L6KDUPQK","created_at":"2026-07-05T07:50:20.000805+00:00"},{"alias_kind":"pith_short_16","alias_value":"4ON4L6KDUPQKSYIC","created_at":"2026-07-05T07:50:20.000805+00:00"},{"alias_kind":"pith_short_8","alias_value":"4ON4L6KD","created_at":"2026-07-05T07:50:20.000805+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2304.08244","citing_title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12944","citing_title":"From Instance Selection to Fixed-Pool Data Recipe Search for Supervised Fine-Tuning","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4ON4L6KDUPQKSYICFQSM4YREMC","json":"https://pith.science/pith/4ON4L6KDUPQKSYICFQSM4YREMC.json","graph_json":"https://pith.science/api/pith-number/4ON4L6KDUPQKSYICFQSM4YREMC/graph.json","events_json":"https://pith.science/api/pith-number/4ON4L6KDUPQKSYICFQSM4YREMC/events.json","paper":"https://pith.science/paper/4ON4L6KD"},"agent_actions":{"view_html":"https://pith.science/pith/4ON4L6KDUPQKSYICFQSM4YREMC","download_json":"https://pith.science/pith/4ON4L6KDUPQKSYICFQSM4YREMC.json","view_paper":"https://pith.science/paper/4ON4L6KD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.05696&json=true","fetch_graph":"https://pith.science/api/pith-number/4ON4L6KDUPQKSYICFQSM4YREMC/graph.json","fetch_events":"https://pith.science/api/pith-number/4ON4L6KDUPQKSYICFQSM4YREMC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4ON4L6KDUPQKSYICFQSM4YREMC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4ON4L6KDUPQKSYICFQSM4YREMC/action/storage_attestation","attest_author":"https://pith.science/pith/4ON4L6KDUPQKSYICFQSM4YREMC/action/author_attestation","sign_citation":"https://pith.science/pith/4ON4L6KDUPQKSYICFQSM4YREMC/action/citation_signature","submit_replication":"https://pith.science/pith/4ON4L6KDUPQKSYICFQSM4YREMC/action/replication_record"}},"created_at":"2026-07-05T07:50:20.000805+00:00","updated_at":"2026-07-05T07:50:20.000805+00:00"}