{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3OE6BOTKHUI75N7VUSGFVMNSIK","short_pith_number":"pith:3OE6BOTK","schema_version":"1.0","canonical_sha256":"db89e0ba6a3d11feb7f5a48c5ab1b2429d239da2ccf3151d694c4389fef7795d","source":{"kind":"arxiv","id":"2507.16792","version":1},"attestation_state":"computed","paper":{"title":"ChatChecker: A Framework for Dialogue System Testing and Evaluation Through Non-cooperative User Simulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Michel Schimpf, Roman Mayr, Thomas Bohn\\'e","submitted_at":"2025-07-22T17:40:34Z","abstract_excerpt":"While modern dialogue systems heavily rely on large language models (LLMs), their implementation often goes beyond pure LLM interaction. Developers integrate multiple LLMs, external tools, and databases. Therefore, assessment of the underlying LLM alone does not suffice, and the dialogue systems must be tested and evaluated as a whole. However, this remains a major challenge. With most previous work focusing on turn-level analysis, less attention has been paid to integrated dialogue-level quality assurance. To address this, we present ChatChecker, a framework for automated evaluation and testi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.16792","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-07-22T17:40:34Z","cross_cats_sorted":[],"title_canon_sha256":"43102ba16c4ff996ca95e97f7b99ba94a007e655399e21f1d70160a463a7870c","abstract_canon_sha256":"c192970f07664a9490fb36bced35bc4796c57a7e8b46f260bb71cfcdb3550134"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:41:31.228738Z","signature_b64":"cMMGZ8njLptMHWBylZuZV64wqpXvbANPy0Tf4mSUzHpW3oXyro04MFYtw5TQbmPaaq6tnIqv6ECDDnfERsdYCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"db89e0ba6a3d11feb7f5a48c5ab1b2429d239da2ccf3151d694c4389fef7795d","last_reissued_at":"2026-07-05T11:41:31.228329Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:41:31.228329Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChatChecker: A Framework for Dialogue System Testing and Evaluation Through Non-cooperative User Simulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Michel Schimpf, Roman Mayr, Thomas Bohn\\'e","submitted_at":"2025-07-22T17:40:34Z","abstract_excerpt":"While modern dialogue systems heavily rely on large language models (LLMs), their implementation often goes beyond pure LLM interaction. Developers integrate multiple LLMs, external tools, and databases. Therefore, assessment of the underlying LLM alone does not suffice, and the dialogue systems must be tested and evaluated as a whole. However, this remains a major challenge. With most previous work focusing on turn-level analysis, less attention has been paid to integrated dialogue-level quality assurance. To address this, we present ChatChecker, a framework for automated evaluation and testi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.16792","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.16792/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.16792","created_at":"2026-07-05T11:41:31.228384+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.16792v1","created_at":"2026-07-05T11:41:31.228384+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.16792","created_at":"2026-07-05T11:41:31.228384+00:00"},{"alias_kind":"pith_short_12","alias_value":"3OE6BOTKHUI7","created_at":"2026-07-05T11:41:31.228384+00:00"},{"alias_kind":"pith_short_16","alias_value":"3OE6BOTKHUI75N7V","created_at":"2026-07-05T11:41:31.228384+00:00"},{"alias_kind":"pith_short_8","alias_value":"3OE6BOTK","created_at":"2026-07-05T11:41:31.228384+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3OE6BOTKHUI75N7VUSGFVMNSIK","json":"https://pith.science/pith/3OE6BOTKHUI75N7VUSGFVMNSIK.json","graph_json":"https://pith.science/api/pith-number/3OE6BOTKHUI75N7VUSGFVMNSIK/graph.json","events_json":"https://pith.science/api/pith-number/3OE6BOTKHUI75N7VUSGFVMNSIK/events.json","paper":"https://pith.science/paper/3OE6BOTK"},"agent_actions":{"view_html":"https://pith.science/pith/3OE6BOTKHUI75N7VUSGFVMNSIK","download_json":"https://pith.science/pith/3OE6BOTKHUI75N7VUSGFVMNSIK.json","view_paper":"https://pith.science/paper/3OE6BOTK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.16792&json=true","fetch_graph":"https://pith.science/api/pith-number/3OE6BOTKHUI75N7VUSGFVMNSIK/graph.json","fetch_events":"https://pith.science/api/pith-number/3OE6BOTKHUI75N7VUSGFVMNSIK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3OE6BOTKHUI75N7VUSGFVMNSIK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3OE6BOTKHUI75N7VUSGFVMNSIK/action/storage_attestation","attest_author":"https://pith.science/pith/3OE6BOTKHUI75N7VUSGFVMNSIK/action/author_attestation","sign_citation":"https://pith.science/pith/3OE6BOTKHUI75N7VUSGFVMNSIK/action/citation_signature","submit_replication":"https://pith.science/pith/3OE6BOTKHUI75N7VUSGFVMNSIK/action/replication_record"}},"created_at":"2026-07-05T11:41:31.228384+00:00","updated_at":"2026-07-05T11:41:31.228384+00:00"}