{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QFI72X4Y4DAMJWST5A7GDZR5WS","short_pith_number":"pith:QFI72X4Y","schema_version":"1.0","canonical_sha256":"8151fd5f98e0c0c4da53e83e61e63db4a5fe424efb4c8af41ab5560bd8ef042d","source":{"kind":"arxiv","id":"2311.09829","version":1},"attestation_state":"computed","paper":{"title":"FollowEval: A Multi-Dimensional Benchmark for Assessing the Instruction-Following Capability of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Deyi Xiong, Huishi Qiu, Jiahao Hu, Peng Wang, Renren Jin, Xiaohua Wang, Yimin Jing","submitted_at":"2023-11-16T11:53:31Z","abstract_excerpt":"The effective assessment of the instruction-following ability of large language models (LLMs) is of paramount importance. A model that cannot adhere to human instructions might be not able to provide reliable and helpful responses. In pursuit of this goal, various benchmarks have been constructed to evaluate the instruction-following capacity of these models. However, these benchmarks are limited to a single language and are constructed using automated approaches, which restricts their applicability and the quality of the test examples they contain. To bridge this gap, we introduce the FollowE"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09829","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-16T11:53:31Z","cross_cats_sorted":[],"title_canon_sha256":"6a9ca080762371e82829f555c5a74ef14ea46fba809367ed3117dd120f9476fe","abstract_canon_sha256":"d2a9ab27d3efc3d11b6314fd09d4a1eb31226ac14811867c173e20f2cac94684"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:13:33.463211Z","signature_b64":"aP3O3qFK3bsWJvRysQR7Z/BI436Lh5U1FmJJ2RSQoawALcCd/PTgYzILa7VUSSrC2xamDV9DeNtBw9Ug1ifBCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8151fd5f98e0c0c4da53e83e61e63db4a5fe424efb4c8af41ab5560bd8ef042d","last_reissued_at":"2026-07-05T07:13:33.462732Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:13:33.462732Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FollowEval: A Multi-Dimensional Benchmark for Assessing the Instruction-Following Capability of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Deyi Xiong, Huishi Qiu, Jiahao Hu, Peng Wang, Renren Jin, Xiaohua Wang, Yimin Jing","submitted_at":"2023-11-16T11:53:31Z","abstract_excerpt":"The effective assessment of the instruction-following ability of large language models (LLMs) is of paramount importance. A model that cannot adhere to human instructions might be not able to provide reliable and helpful responses. In pursuit of this goal, various benchmarks have been constructed to evaluate the instruction-following capacity of these models. However, these benchmarks are limited to a single language and are constructed using automated approaches, which restricts their applicability and the quality of the test examples they contain. To bridge this gap, we introduce the FollowE"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09829","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09829/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09829","created_at":"2026-07-05T07:13:33.462792+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09829v1","created_at":"2026-07-05T07:13:33.462792+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09829","created_at":"2026-07-05T07:13:33.462792+00:00"},{"alias_kind":"pith_short_12","alias_value":"QFI72X4Y4DAM","created_at":"2026-07-05T07:13:33.462792+00:00"},{"alias_kind":"pith_short_16","alias_value":"QFI72X4Y4DAMJWST","created_at":"2026-07-05T07:13:33.462792+00:00"},{"alias_kind":"pith_short_8","alias_value":"QFI72X4Y","created_at":"2026-07-05T07:13:33.462792+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.11290","citing_title":"ReAD: Reinforcement-Guided Capability Distillation for Large Language Models","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QFI72X4Y4DAMJWST5A7GDZR5WS","json":"https://pith.science/pith/QFI72X4Y4DAMJWST5A7GDZR5WS.json","graph_json":"https://pith.science/api/pith-number/QFI72X4Y4DAMJWST5A7GDZR5WS/graph.json","events_json":"https://pith.science/api/pith-number/QFI72X4Y4DAMJWST5A7GDZR5WS/events.json","paper":"https://pith.science/paper/QFI72X4Y"},"agent_actions":{"view_html":"https://pith.science/pith/QFI72X4Y4DAMJWST5A7GDZR5WS","download_json":"https://pith.science/pith/QFI72X4Y4DAMJWST5A7GDZR5WS.json","view_paper":"https://pith.science/paper/QFI72X4Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09829&json=true","fetch_graph":"https://pith.science/api/pith-number/QFI72X4Y4DAMJWST5A7GDZR5WS/graph.json","fetch_events":"https://pith.science/api/pith-number/QFI72X4Y4DAMJWST5A7GDZR5WS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QFI72X4Y4DAMJWST5A7GDZR5WS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QFI72X4Y4DAMJWST5A7GDZR5WS/action/storage_attestation","attest_author":"https://pith.science/pith/QFI72X4Y4DAMJWST5A7GDZR5WS/action/author_attestation","sign_citation":"https://pith.science/pith/QFI72X4Y4DAMJWST5A7GDZR5WS/action/citation_signature","submit_replication":"https://pith.science/pith/QFI72X4Y4DAMJWST5A7GDZR5WS/action/replication_record"}},"created_at":"2026-07-05T07:13:33.462792+00:00","updated_at":"2026-07-05T07:13:33.462792+00:00"}