{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TVXGGDSAKBZNZ6GMNVBWTO6WWZ","short_pith_number":"pith:TVXGGDSA","schema_version":"1.0","canonical_sha256":"9d6e630e405072dcf8cc6d4369bbd6b66694e422ae8058147ff7c76a19797bf7","source":{"kind":"arxiv","id":"2311.01767","version":2},"attestation_state":"computed","paper":{"title":"PPTC Benchmark: Evaluating Large Language Models for PowerPoint Task Completion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dongyan Zhao, Nan Duan, Yaobo Liang, Yiduo Guo, Zekai Zhang","submitted_at":"2023-11-03T08:06:35Z","abstract_excerpt":"Recent evaluations of Large Language Models (LLMs) have centered around testing their zero-shot/few-shot capabilities for basic natural language tasks and their ability to translate instructions into tool APIs. However, the evaluation of LLMs utilizing complex tools to finish multi-turn, multi-modal instructions in a complex multi-modal environment has not been investigated. To address this gap, we introduce the PowerPoint Task Completion (PPTC) benchmark to assess LLMs' ability to create and edit PPT files based on user instructions. It contains 279 multi-turn sessions covering diverse topics"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.01767","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-03T08:06:35Z","cross_cats_sorted":[],"title_canon_sha256":"b09a606045400d4a8352e9ccf8223660338ce420b877aaaf803a7682966acfed","abstract_canon_sha256":"c901d91c5bb15a9ef0398b42623f70a47e6370a946cdae7212438630d59b94e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:09:56.084553Z","signature_b64":"yI/zepkThKPU8nnmZyx/3bZEWOl/U/wMAh+8Kgic4QoZDTJbi/3JDp4wRppQfT+XyeU4b11s2Igc5JjmfOoIBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d6e630e405072dcf8cc6d4369bbd6b66694e422ae8058147ff7c76a19797bf7","last_reissued_at":"2026-07-05T07:09:56.084057Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:09:56.084057Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PPTC Benchmark: Evaluating Large Language Models for PowerPoint Task Completion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dongyan Zhao, Nan Duan, Yaobo Liang, Yiduo Guo, Zekai Zhang","submitted_at":"2023-11-03T08:06:35Z","abstract_excerpt":"Recent evaluations of Large Language Models (LLMs) have centered around testing their zero-shot/few-shot capabilities for basic natural language tasks and their ability to translate instructions into tool APIs. However, the evaluation of LLMs utilizing complex tools to finish multi-turn, multi-modal instructions in a complex multi-modal environment has not been investigated. To address this gap, we introduce the PowerPoint Task Completion (PPTC) benchmark to assess LLMs' ability to create and edit PPT files based on user instructions. It contains 279 multi-turn sessions covering diverse topics"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.01767","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.01767/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.01767","created_at":"2026-07-05T07:09:56.084111+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.01767v2","created_at":"2026-07-05T07:09:56.084111+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.01767","created_at":"2026-07-05T07:09:56.084111+00:00"},{"alias_kind":"pith_short_12","alias_value":"TVXGGDSAKBZN","created_at":"2026-07-05T07:09:56.084111+00:00"},{"alias_kind":"pith_short_16","alias_value":"TVXGGDSAKBZNZ6GM","created_at":"2026-07-05T07:09:56.084111+00:00"},{"alias_kind":"pith_short_8","alias_value":"TVXGGDSA","created_at":"2026-07-05T07:09:56.084111+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.23218","citing_title":"OS-ATLAS: A Foundation Action Model for Generalist GUI Agents","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07972","citing_title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TVXGGDSAKBZNZ6GMNVBWTO6WWZ","json":"https://pith.science/pith/TVXGGDSAKBZNZ6GMNVBWTO6WWZ.json","graph_json":"https://pith.science/api/pith-number/TVXGGDSAKBZNZ6GMNVBWTO6WWZ/graph.json","events_json":"https://pith.science/api/pith-number/TVXGGDSAKBZNZ6GMNVBWTO6WWZ/events.json","paper":"https://pith.science/paper/TVXGGDSA"},"agent_actions":{"view_html":"https://pith.science/pith/TVXGGDSAKBZNZ6GMNVBWTO6WWZ","download_json":"https://pith.science/pith/TVXGGDSAKBZNZ6GMNVBWTO6WWZ.json","view_paper":"https://pith.science/paper/TVXGGDSA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.01767&json=true","fetch_graph":"https://pith.science/api/pith-number/TVXGGDSAKBZNZ6GMNVBWTO6WWZ/graph.json","fetch_events":"https://pith.science/api/pith-number/TVXGGDSAKBZNZ6GMNVBWTO6WWZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TVXGGDSAKBZNZ6GMNVBWTO6WWZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TVXGGDSAKBZNZ6GMNVBWTO6WWZ/action/storage_attestation","attest_author":"https://pith.science/pith/TVXGGDSAKBZNZ6GMNVBWTO6WWZ/action/author_attestation","sign_citation":"https://pith.science/pith/TVXGGDSAKBZNZ6GMNVBWTO6WWZ/action/citation_signature","submit_replication":"https://pith.science/pith/TVXGGDSAKBZNZ6GMNVBWTO6WWZ/action/replication_record"}},"created_at":"2026-07-05T07:09:56.084111+00:00","updated_at":"2026-07-05T07:09:56.084111+00:00"}