{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IPCDUZFD2CQMOEQILE5ZKVWJJA","short_pith_number":"pith:IPCDUZFD","schema_version":"1.0","canonical_sha256":"43c43a64a3d0a0c71208593b9556c9481bbaa8cae9d0b0bbe7c2a01a94a81e25","source":{"kind":"arxiv","id":"2311.09184","version":2},"attestation_state":"computed","paper":{"title":"Benchmarking Generation and Evaluation Capabilities of Large Language Models for Instruction Controllable Summarization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander R. Fabbri, Arman Cohan, Chien-Sheng Wu, Dragomir Radev, Jiawen Chen, Pengfei Liu, Shafiq Joty, Simeng Han, Yilun Zhao, Yixin Liu","submitted_at":"2023-11-15T18:25:26Z","abstract_excerpt":"While large language models (LLMs) can already achieve strong performance on standard generic summarization benchmarks, their performance on more complex summarization task settings is less studied. Therefore, we benchmark LLMs on instruction controllable text summarization, where the model input consists of both a source article and a natural language requirement for desired summary characteristics. To this end, we curate an evaluation-only dataset for this task setting and conduct human evaluations of five LLM-based systems to assess their instruction-following capabilities in controllable s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09184","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-15T18:25:26Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"19d0174f75e58f613c415fc8cc2b051f4fe8722871c4e199727a344168ee4cb4","abstract_canon_sha256":"3e9efecf7340d58dc714c576f157b1341cf101ff7e033d1227deb259759d55c5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:07.644086Z","signature_b64":"n4JDoKwFwP8h8YBcak3AywIF56NK63vsbkt2BdG6lhLk+qhqqpvGLoVAWZvdEJiwJh08F7HamA9+qtbqLXXYDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43c43a64a3d0a0c71208593b9556c9481bbaa8cae9d0b0bbe7c2a01a94a81e25","last_reissued_at":"2026-07-05T08:43:07.643619Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:07.643619Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Generation and Evaluation Capabilities of Large Language Models for Instruction Controllable Summarization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander R. Fabbri, Arman Cohan, Chien-Sheng Wu, Dragomir Radev, Jiawen Chen, Pengfei Liu, Shafiq Joty, Simeng Han, Yilun Zhao, Yixin Liu","submitted_at":"2023-11-15T18:25:26Z","abstract_excerpt":"While large language models (LLMs) can already achieve strong performance on standard generic summarization benchmarks, their performance on more complex summarization task settings is less studied. Therefore, we benchmark LLMs on instruction controllable text summarization, where the model input consists of both a source article and a natural language requirement for desired summary characteristics. To this end, we curate an evaluation-only dataset for this task setting and conduct human evaluations of five LLM-based systems to assess their instruction-following capabilities in controllable s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09184","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09184/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09184","created_at":"2026-07-05T08:43:07.643677+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09184v2","created_at":"2026-07-05T08:43:07.643677+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09184","created_at":"2026-07-05T08:43:07.643677+00:00"},{"alias_kind":"pith_short_12","alias_value":"IPCDUZFD2CQM","created_at":"2026-07-05T08:43:07.643677+00:00"},{"alias_kind":"pith_short_16","alias_value":"IPCDUZFD2CQMOEQI","created_at":"2026-07-05T08:43:07.643677+00:00"},{"alias_kind":"pith_short_8","alias_value":"IPCDUZFD","created_at":"2026-07-05T08:43:07.643677+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2405.14782","citing_title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","ref_index":154,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IPCDUZFD2CQMOEQILE5ZKVWJJA","json":"https://pith.science/pith/IPCDUZFD2CQMOEQILE5ZKVWJJA.json","graph_json":"https://pith.science/api/pith-number/IPCDUZFD2CQMOEQILE5ZKVWJJA/graph.json","events_json":"https://pith.science/api/pith-number/IPCDUZFD2CQMOEQILE5ZKVWJJA/events.json","paper":"https://pith.science/paper/IPCDUZFD"},"agent_actions":{"view_html":"https://pith.science/pith/IPCDUZFD2CQMOEQILE5ZKVWJJA","download_json":"https://pith.science/pith/IPCDUZFD2CQMOEQILE5ZKVWJJA.json","view_paper":"https://pith.science/paper/IPCDUZFD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09184&json=true","fetch_graph":"https://pith.science/api/pith-number/IPCDUZFD2CQMOEQILE5ZKVWJJA/graph.json","fetch_events":"https://pith.science/api/pith-number/IPCDUZFD2CQMOEQILE5ZKVWJJA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IPCDUZFD2CQMOEQILE5ZKVWJJA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IPCDUZFD2CQMOEQILE5ZKVWJJA/action/storage_attestation","attest_author":"https://pith.science/pith/IPCDUZFD2CQMOEQILE5ZKVWJJA/action/author_attestation","sign_citation":"https://pith.science/pith/IPCDUZFD2CQMOEQILE5ZKVWJJA/action/citation_signature","submit_replication":"https://pith.science/pith/IPCDUZFD2CQMOEQILE5ZKVWJJA/action/replication_record"}},"created_at":"2026-07-05T08:43:07.643677+00:00","updated_at":"2026-07-05T08:43:07.643677+00:00"}