{"as_of":"2026-08-20T01:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c84462d43e55eff67e421a41ebec89a142e948b20ae9ba99d50b3e518e41ac62","coverage":[{"denominator":66,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T23:08:12.028546Z","state":"measured"},{"denominator":66,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":66,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.05445/citation-record","integrity":"/paper/2505.05445/integrity","json":"/paper/2505.05445/citation-record.json","paper":"/paper/2505.05445"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.438572Z","title":null,"venue":null,"work_id":"817e0b92-520e-417d-9741-43db4cfa7189","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.549825Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:e310904d248dd4c528a814c33ef49730c1fdb3698d241ea51eacba48d71b9500","observation_id":"907bed5d-a6bf-442f-80b1-334a62169694","resolution":{"observed_at":"2026-08-15T23:08:13.446089Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.416071Z","title":null,"venue":null,"work_id":"528563b6-3d39-4147-9a67-8b9606bde31e","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.557225Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:d9a285ad9cfb440352b017a682c99089f45ce18b30dc07eff93e3193eb87df79","observation_id":"34ccef67-43d4-458d-aef5-344a80638106","resolution":{"observed_at":"2026-08-15T23:08:13.423542Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.396203Z","title":null,"venue":null,"work_id":"94d1c120-1595-48a1-beba-53fec7aab22f","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.565179Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:1dee9174b4239ed4a47d18c9c69fc5a154f31b9b46d5a60b7fee82cfbc9fc1f4","observation_id":"5fb3cfe9-1840-4273-b8d7-04a06328507d","resolution":{"observed_at":"2026-08-15T23:08:13.403762Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00427","last_updated":"2024-11-01T07:50:19Z","snapshot_observed_at":"2026-08-19T22:10:34.649102Z","submitted_at":"2024-11-01T07:50:19Z","title":"DARD: A Multi-Agent Approach for Task-Oriented Dialog Systems","version":1},"cited_work":{"arxiv_id":"2411.00427","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.00427","snapshot_observed_at":"2026-08-15T23:08:12.098924Z","title":"DARD: A Multi-Agent Approach for Task-Oriented Dialog Systems","venue":"cs.CL","work_id":"fe607431-5f89-4d63-9b60-dc8490231b29","year":2024},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.533824Z"},"links":{"cited_paper":"/paper/2411.00427","citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:0ba355f4b4b828a855d5b6d799eaa5d6a6a023add99d5d8249ecf39631b4a533","observation_id":"16639a4a-f265-4bea-ace3-a4bba488f0fc","resolution":{"observed_at":"2026-08-15T23:08:12.107557Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.377204Z","title":"Single” refers to tasks within a single domain (Restaurant(R), Hotel(H), or Train(T)), while “Multi","venue":null,"work_id":"70aad18c-7f66-4865-9695-336f7d82c106","year":2023},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.572474Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:406ffea8d3038ea8fc70de5146e2e854bf91b2b2011ad1dd7010bcddd497e7b4","observation_id":"32394b58-82cc-4ceb-8751-4684d7ba9ba7","resolution":{"observed_at":"2026-08-15T23:08:13.383786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.152986Z","title":"Lets begin Figure 6: Prompt template for the User Simulator, specifying the task description, interaction instructions, and response format guidelines","venue":null,"work_id":"6a2fe004-3668-4dce-a7d9-feebb07dd964","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.661478Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:c1600812d59ea7e9f12227bfed7bf84f06e0b937a76c1c2d739f42640e88201a","observation_id":"ae977ad7-b7bc-40eb-83df-827c94f13726","resolution":{"observed_at":"2026-08-15T23:08:13.160307Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.134864Z","title":null,"venue":null,"work_id":"af55b7fb-7005-4634-8633-43969c19864c","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.669955Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:bdb978cfab804d8f4cf1d464b88f28564adae93dab48f44b927bbe109aea3501","observation_id":"01576dfd-0a10-43ba-9347-37bd7fe5fe31","resolution":{"observed_at":"2026-08-15T23:08:13.140580Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.113803Z","title":null,"venue":null,"work_id":"c8d7af36-2309-404d-9d29-bbd5131dce5c","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.675818Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:53e0c01f694eb60dc2d734dffb864e79c65796f6c1b2933aa2f0927d1490ab22","observation_id":"bd142308-b2ec-440d-b0dd-b1710653b2a0","resolution":{"observed_at":"2026-08-15T23:08:13.123308Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.095010Z","title":"If the required information is not available in the returned records, apply additional filters to narrow down the results","venue":null,"work_id":"a673df2b-2e6e-4d07-8100-36f994061a6b","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.681392Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:1693076831bbb60cb36d14a8718233eddfa5361cc2c539f3adfc56b7e4115714","observation_id":"e7fb71ef-f703-4529-89fb-dee6776a364b","resolution":{"observed_at":"2026-08-15T23:08:13.101004Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.076977Z","title":null,"venue":null,"work_id":"73d57e68-c1ad-4f33-a1b4-a43a6491be21","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.687470Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:51874c259cf5e25378c75a31c510890fcb506db38ff96b236cf8b15db94fe35f","observation_id":"400b6989-a494-442e-84de-3515f89ecfb9","resolution":{"observed_at":"2026-08-15T23:08:13.082232Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.060375Z","title":null,"venue":null,"work_id":"a2a9f70b-4581-4d41-8720-e53ae0160c19","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.693004Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:53d1e9783f79a8ad99c1a1b2492fd43f922914e94d33ea75d5154ac260e883b7","observation_id":"14070b00-38ca-414f-91fd-7861b8861259","resolution":{"observed_at":"2026-08-15T23:08:13.066106Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.042263Z","title":null,"venue":null,"work_id":"117d38c9-0512-48f0-8b5c-734c63468e47","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.699329Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:12d558f10360058431211a7261529d906f6949487330bcdd3347e695dc554652","observation_id":"fcd0f4da-a38c-4252-a089-7d67a3c306c1","resolution":{"observed_at":"2026-08-15T23:08:13.048080Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.024349Z","title":null,"venue":null,"work_id":"c783d470-cec5-4b00-abee-ec9b8a821bfe","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.704573Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:5b27d688331b726aa7a00c6782e61b6691656b1134dc8e025eff631ca68e78f0","observation_id":"ec89b911-89c4-41b6-a6af-c70476bc0de8","resolution":{"observed_at":"2026-08-15T23:08:13.030437Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.005815Z","title":"INSTRUCTIONS:","venue":null,"work_id":"83551975-81ed-405d-8995-ff89ba028760","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.710531Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:aebf654f8e402d5077fe2e35a4ceedcbad7d9220534f58330402dcc9c1cd57f8","observation_id":"9c090d40-11e6-4688-acce-bbfd9db5b6f7","resolution":{"observed_at":"2026-08-15T23:08:13.012016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.357389Z","title":null,"venue":null,"work_id":"d9a57d13-4e98-44e7-a765-127ba814a01c","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.717521Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:8cfdd260926d3eba9d141464caa7a63c75ad2dd540bea65108085377d5bb1dca","observation_id":"22a37baf-93f0-4aab-bd26-457990d418c0","resolution":{"observed_at":"2026-08-15T23:08:13.364401Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.337039Z","title":null,"venue":null,"work_id":"c34d0d62-4ed8-4492-9ba1-3f6d041f458e","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.723470Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:6821e8c3399a60701489ca1cf65b93b3a17cb1939e20598d7573db854e01c08e","observation_id":"9b975fe2-d00c-4a58-ad97-30639acb2afe","resolution":{"observed_at":"2026-08-15T23:08:13.343234Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.316137Z","title":"Do not add logic or interpretation beyond what is explicitly stated in the TASK","venue":null,"work_id":"73cdfc4d-7c99-4732-962a-489e17bbbb45","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.728729Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:d8b8056becd2300c086f368b87aa20f43d84b28608bf1f89d0b7189087eb97f3","observation_id":"6c208875-810b-43e1-b2a4-2984bf5c57d6","resolution":{"observed_at":"2026-08-15T23:08:13.323235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.297072Z","title":"Maintain strict adherence to the original formatting","venue":null,"work_id":"bf083e01-3838-4c24-a45b-4403ece62f21","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.735492Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:35332c5b2d7a3c9893bb4fefc65edb598ec5d85cdb7dd099862f6858423cfba3","observation_id":"c5860fd8-b854-443c-bee7-917025cbe44b","resolution":{"observed_at":"2026-08-15T23:08:13.303018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.278044Z","title":"If the dialogue system provides alternative options close to the desired time, the you should accept a suitable nearby option that reasonably aligns with the goal","venue":null,"work_id":"9e8295c2-2f6a-4366-9943-adc7afcb3738","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.742347Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:b746918a316721d47f2231ad82d1b74172bc244d1979be0c52b7264d8f5d7295","observation_id":"db968260-6f71-43d7-8efc-08da1d6a1540","resolution":{"observed_at":"2026-08-15T23:08:13.283977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.257452Z","title":"No additional text should follow/preceed","venue":null,"work_id":"023515ba-06a4-45a3-8887-fce0ee67e15e","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.747466Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:bd48a0b18be05de5105ea6e02f7ffa7b51556ebbd62e13692cc8244dc3251d08","observation_id":"c44332b8-7092-4c21-9446-876021605188","resolution":{"observed_at":"2026-08-15T23:08:13.263466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.234236Z","title":null,"venue":null,"work_id":"d49ba452-03a3-4ba0-9583-1b46c6179d63","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.754113Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:51ec6ce6c80c6b15143b6da3bb0793024a32ec87db16cac0652588be2168287d","observation_id":"cd806017-94dd-44ee-bc2e-3ca182afff3c","resolution":{"observed_at":"2026-08-15T23:08:13.241304Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.215166Z","title":"OUTPUT FORMAT","venue":null,"work_id":"4b7db038-9964-42fc-a511-917ee11b5c2b","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.759521Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:e7b75d4996e641a98c85aeaab022186e92faebb4e9fc7149eb4e33dc0ead0289","observation_id":"06b26831-bb77-4a94-b4c7-1162f8bbcd08","resolution":{"observed_at":"2026-08-15T23:08:13.221403Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.194973Z","title":null,"venue":null,"work_id":"be65037e-6988-4388-ab00-e6dfec5a1fe2","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.765472Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:9325c276f155a86182a8457eeb4ce7a3d6dd2cb4c440dfa9b01ee0b3663afe2d","observation_id":"70b5ad72-d54e-4664-80c9-05e5302da2f0","resolution":{"observed_at":"2026-08-15T23:08:13.202296Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.172717Z","title":"Do not add any customary comments (thank you, great etc.)","venue":null,"work_id":"5d24a65e-b2f1-4c41-a5e6-88e591a8644a","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.770168Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:4e6a8d79919e60c20681045c3dc4bb1bd2949d9185ad73b0809800c549c84e03","observation_id":"d13b24a0-dbb2-4e88-bd4b-723b6fb7a9d1","resolution":{"observed_at":"2026-08-15T23:08:13.180119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.985414Z","title":null,"venue":null,"work_id":"36190941-80b2-4fcc-93f1-a69902345639","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.775275Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:2618059b5bc80713d480ed76a07e40681113450689005e5c09eef6c99aa1f86f","observation_id":"97a1fa13-528d-41bc-8473-9e389467c42e","resolution":{"observed_at":"2026-08-15T23:08:12.992131Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.964401Z","title":"Determine appropriate flow based on user input and available information b","venue":null,"work_id":"2309244b-3e0c-4377-aab9-5d167e97d113","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.780866Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:be9b7fb791674a69e20ee580d2bb017f81c1125a7e0fd64d9d7a0f652034d93d","observation_id":"07c661ea-b65c-4d31-b474-282536a4cbe1","resolution":{"observed_at":"2026-08-15T23:08:12.971765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.942745Z","title":"Include all required fields and the response must be a valid JSON","venue":null,"work_id":"8cd1311f-5deb-4a29-b32a-83c6365d1328","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.787537Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:99a35d0f7e60223285086c948886c751237af264b79c276af2eafa0a1663eeb3","observation_id":"3c95b3cc-db40-49fb-8dee-0a320b4217b6","resolution":{"observed_at":"2026-08-15T23:08:12.950946Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.893557Z","title":null,"venue":null,"work_id":"60bc2c63-3649-4cc7-a0e8-e46b922dcd38","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.799378Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:41a79d6634fdb31acccf042031a04ef96d33484e54b302dc7023978f09497083","observation_id":"5c13c4d8-82f8-4257-ad87-f43169a347dd","resolution":{"observed_at":"2026-08-15T23:08:12.901429Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.872917Z","title":null,"venue":null,"work_id":"db992f28-c97f-4bc8-b47b-fdc9d795482a","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.804730Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:9300873a0453297503f6128356a0daa5651d7888b78f490a0622875275b11e04","observation_id":"476411b6-5365-410e-81ea-beb2e7655030","resolution":{"observed_at":"2026-08-15T23:08:12.879017Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.850812Z","title":null,"venue":null,"work_id":"d3b8054a-42f3-45d9-a73f-186d62fd2439","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.809945Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:7c17b2c272a186bf4dd8270d08cc756f6fda6b2d577a42a36b9e052515d868fc","observation_id":"b93d7e9e-fc37-4a76-bfbf-4f98f62204fb","resolution":{"observed_at":"2026-08-15T23:08:12.858032Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.805921Z","title":"Never return multiple function calls in a single response","venue":null,"work_id":"8f5c14ab-8a08-4add-8017-71b8b5f4b514","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.820654Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:228eb68874df9b2117d0e2d2d2b51909dedc8a73cf048ff6bb860281a1949cc3","observation_id":"581c50c5-49a5-446c-aa86-82302916949a","resolution":{"observed_at":"2026-08-15T23:08:12.812548Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.785218Z","title":"The task is completed ONLY if all the intentions are fulfilled","venue":null,"work_id":"2b9c36ce-bc35-43f0-b3e9-83ba5e31b30a","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.826846Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:d761b71dc896ac359112ced8f642e58424ff9365a6790b908323e8fd19933561","observation_id":"7c9be236-8f5b-40df-910a-b6447d75970e","resolution":{"observed_at":"2026-08-15T23:08:12.792357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.764538Z","title":"In the dialogue, the user or the system could either be AI or human","venue":null,"work_id":"e2215b26-a6b4-4feb-953c-92c3ba1c1c4f","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.833574Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:612352331c7a2bd2beabc90541ba8bd9aef50bbc1371cb3afa817ab54095be13","observation_id":"68c7aa03-5380-45d0-86d7-4a24518ed800","resolution":{"observed_at":"2026-08-15T23:08:12.771712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.742890Z","title":"You should report a numeric rating from 1 to 3, where 3 represents the best coherence","venue":null,"work_id":"4722ea1c-2222-4636-bb4d-2315d676e60b","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.841174Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:d7d57e5e039a94ef1e89488bc8f156dcc08ceac22967c88caf334035e9514633","observation_id":"f3fe30f7-8c56-459f-add3-ea87b54399db","resolution":{"observed_at":"2026-08-15T23:08:12.748698Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.720771Z","title":"For dialogue-level diversity, you only need to evaluate the user","venue":null,"work_id":"a922c364-d975-4fec-8030-9526cf35e1b1","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.847118Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:1202396355d278ebefaae2cd9c43437cc89b0defdba6ca2a756e14a4e081c67f","observation_id":"ef18a07d-f1a9-45fd-8f10-93f9246d89dc","resolution":{"observed_at":"2026-08-15T23:08:12.728888Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.697477Z","title":null,"venue":null,"work_id":"702083a1-62fd-4c5a-91e6-2044f765b1d7","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.852987Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:9123516ce027b932d7bf1ddb93d8978fc0d4f078b74412994da2616887cbaacd","observation_id":"4ceb158b-36f8-47c9-81fe-4f346d54fc85","resolution":{"observed_at":"2026-08-15T23:08:12.704956Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.676306Z","title":null,"venue":null,"work_id":"0793868e-9825-4780-b150-ec2ceecda742","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.859434Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:b020007a1b0acf14c1f83b18a11ac8d1769e8f34622b7038b96b4a122e4b33b8","observation_id":"cb0362ea-0dd5-46d6-a4cb-d6edb88e01e0","resolution":{"observed_at":"2026-08-15T23:08:12.683535Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.652688Z","title":null,"venue":null,"work_id":"20499d36-ad3e-496e-9f9c-53027edbc334","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.864691Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:003a0ecfb744f1c8b2753b036a76bea508d0c06b564fc1621e9127c7825eb751","observation_id":"4a092fcd-d86d-4475-8b77-f90ea8c25d55","resolution":{"observed_at":"2026-08-15T23:08:12.660118Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.621451Z","title":null,"venue":null,"work_id":"1505e12c-77c3-469f-91dc-cbd2d2fb8281","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.870191Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:8a612aaef916c90e8ca30a66a3315257cb3756cb5f1f62db1cc8664ba0b14b0f","observation_id":"dcecedd9-f3dd-4fe2-8acf-d1bcf955b0fb","resolution":{"observed_at":"2026-08-15T23:08:12.630827Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.600755Z","title":null,"venue":null,"work_id":"c450d36d-3165-48ca-8eeb-babd10b8f152","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.876084Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:c9d3b60f55b28c5b12913601113058eb01bc2b20bdfab31b124bd08f3f5b24c1","observation_id":"ccba7b6c-0166-40ad-9156-7dd034be9cc3","resolution":{"observed_at":"2026-08-15T23:08:12.607919Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.581750Z","title":null,"venue":null,"work_id":"46107b18-0fc3-4e3d-90f4-16017ae5f503","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.881835Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:10edb9ad6ed532209a3b484e601b1c9297e8f5c229f51f5032bcda377e374ea2","observation_id":"08c8aeac-bad8-4b99-9fae-d1ccce2b053e","resolution":{"observed_at":"2026-08-15T23:08:12.588037Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.558291Z","title":null,"venue":null,"work_id":"28ea1793-753f-4d9c-8e04-9f8be022264a","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.889348Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:c42a42c4b76a795590bcc54a7010b7d752db8e75c2945ef6261c344c75b3ef91","observation_id":"26307476-c901-4cca-a601-2ae745c6384d","resolution":{"observed_at":"2026-08-15T23:08:12.566805Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.530472Z","title":null,"venue":null,"work_id":"8e5ff7df-1ca4-4205-b301-1bb19542443d","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.894397Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:b11f4f8f127c28dc2ac8d5517ea20b2f4c5e9a9bd7dbd8c09f0ff6c22a6884b7","observation_id":"fcf601ce-8923-48cf-8509-0f31d7360d20","resolution":{"observed_at":"2026-08-15T23:08:12.536473Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.508498Z","title":"donotcare","venue":null,"work_id":"90128976-d698-479b-92f3-16e1866f494a","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.900560Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:e75299379e80b24a65e9ccab11991c8ab73866c3705242cefda9a12c1e9eba4d","observation_id":"53a807a2-e8e0-4315-9bf0-d569e7a195ae","resolution":{"observed_at":"2026-08-15T23:08:12.514855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.480909Z","title":"Dialogue history is provided to understand the context better","venue":null,"work_id":"8adab950-4009-4743-8061-c4a00c5b1f74","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.905816Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:e446c82f0a17bc4ff398a04b38e387808967825a5d6edd6e3d7a800c3f9cf50e","observation_id":"bcbe4ba1-8370-4040-804f-463f03d35622","resolution":{"observed_at":"2026-08-15T23:08:12.487757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.458197Z","title":null,"venue":null,"work_id":"0b43b57d-1e59-447b-ab85-56f3123e40ff","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.910921Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:19d88cd5279391a162d7f27402ec36b271e048d23aa68668d0e7fa4de953ab44","observation_id":"33216a33-63b0-4a9f-afbe-bc75a9fe17aa","resolution":{"observed_at":"2026-08-15T23:08:12.466774Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.435530Z","title":null,"venue":null,"work_id":"c6587dbc-9888-4137-9445-e95d32c178ac","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.915875Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:99bebeb3b3ee64815290cf0d016e86832a97e8c477a0b0bec72366ac7f24dedf","observation_id":"daacb79d-e931-4c82-9abb-636af78df642","resolution":{"observed_at":"2026-08-15T23:08:12.442625Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.413931Z","title":null,"venue":null,"work_id":"22f088ff-047c-44ba-835d-70c7214b0a47","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.921090Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:8a2d3fcfd72586b18661f8b0df424aaf19a89e620967a2869f476e0019d10645","observation_id":"6fee7c14-4e68-41d5-b335-4a90890516a4","resolution":{"observed_at":"2026-08-15T23:08:12.421182Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.392320Z","title":null,"venue":null,"work_id":"94bd28c8-028f-42ea-abac-22001bb6d83c","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.932498Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:25e91f3e755c6e3b2381cf0ba43e0d593ec7310685c9c1c35abcd9958861d380","observation_id":"27358541-de90-459b-88ae-57543b934739","resolution":{"observed_at":"2026-08-15T23:08:12.399229Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.371361Z","title":null,"venue":null,"work_id":"21a68de5-b275-41cc-9313-8e7b544a75ed","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.940850Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:09040547f8a6a9229115799d4b1afe845252d26985aaf6b13431cd5cb04c2f72","observation_id":"16d92e8e-6636-4500-bf5c-8bee7ffdcf45","resolution":{"observed_at":"2026-08-15T23:08:12.377040Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.351734Z","title":null,"venue":null,"work_id":"1e028291-bc5a-4730-82e1-ac82289ee449","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.951647Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:f9db200670e656a03823bd55c1f60cb92e9a701491aed3d12a34237f08d8be6c","observation_id":"46490a4a-45b4-4361-b2d4-339734b1f43d","resolution":{"observed_at":"2026-08-15T23:08:12.358564Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.329130Z","title":null,"venue":null,"work_id":"85a29f5c-2b83-4be0-931d-49211cda6fe7","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.958834Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:d3116d30902e7be15801276c48389e04a9340c7b133a36bdb6517cd535cb63a8","observation_id":"a9d14941-4a0f-487e-887b-c2cab36f2b77","resolution":{"observed_at":"2026-08-15T23:08:12.337299Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.306440Z","title":null,"venue":null,"work_id":"c7dd0091-cb40-4acb-b4e4-4edd95f6190b","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.966232Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:4272ab0b8aaaf73cb10db4a5a226b1e9de6dc53a42f9a1068633947544761ef8","observation_id":"d1ce788f-bf88-4e1a-8ac2-e4cd63288316","resolution":{"observed_at":"2026-08-15T23:08:12.312769Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.285857Z","title":null,"venue":null,"work_id":"408866e3-761f-4c12-8f89-434e39322987","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.972973Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:9dfb1574390087b28517540a8df4b364dfdcf1811541a80eabce45c2fc3162a5","observation_id":"be36b129-bdcc-4a4a-a2a8-3bc6b95661b2","resolution":{"observed_at":"2026-08-15T23:08:12.292363Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.261776Z","title":"Do not infer, assume, or hallucinate values based on common patterns or prior examples","venue":null,"work_id":"05022e63-4ff3-4182-b7d4-78588aad618d","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.980091Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:9af77dce6198d3b5a68f81fa0c5a76557fd56fc32011a07e18c9af609682f96d","observation_id":"e9264cdb-a551-4d12-914d-15a7011b5649","resolution":{"observed_at":"2026-08-15T23:08:12.269886Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.241180Z","title":null,"venue":null,"work_id":"18a479b9-f769-4d82-bef6-1f041ba9609e","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.987963Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:e75b04b33c151aa540bb5fa8a6687eea01a3e788cfb210663dab168929dd17d1","observation_id":"3ed15598-4dd1-44c3-8db2-d6757b6828e6","resolution":{"observed_at":"2026-08-15T23:08:12.247730Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.222818Z","title":"area\": \"","venue":null,"work_id":"1af929d3-5331-46d2-8296-1551b387bb30","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.996719Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:525bdd02793fe63e2699f942849311acc834e0f25a52881790912c14868ccd97","observation_id":"8a08f849-167d-4b66-b691-f7efc88c957e","resolution":{"observed_at":"2026-08-15T23:08:12.228769Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.915572Z","title":null,"venue":null,"work_id":"99c4c819-c41a-4cac-a6c8-a0a77ae64631","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:12.002912Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:d347e58698f984bf407560ece30d054ae541121fc178cc2df68a195d2cd58b4a","observation_id":"3d5f2287-afab-475d-80b7-9a56b2ef5328","resolution":{"observed_at":"2026-08-15T23:08:12.923185Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.825748Z","title":"Never respond with plain text","venue":null,"work_id":"1abef250-ef3a-4457-b586-8f983626b3c0","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:12.009465Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:f97df49e2213afd86a3c64337711a4d2c4dbfcd46e11b02ebd7a87b1a5ceabbc","observation_id":"2482252d-48f8-4021-9352-bbebacdfe242","resolution":{"observed_at":"2026-08-15T23:08:12.836983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.203763Z","title":null,"venue":null,"work_id":"e60d3b27-f76c-4714-bbe3-6096c3d1cd5a","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:12.015038Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:b954b5979316f8183eec12d8eafd785955e3ef9bcc2f6df2cd257e3c7b0b5319","observation_id":"717f6d3e-23cc-4578-9116-b0a4585a7b4a","resolution":{"observed_at":"2026-08-15T23:08:12.210076Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.182994Z","title":"If additional information is required to proceed, respond conversationally using direct and focused phrasing","venue":null,"work_id":"41869127-e5d4-4945-b206-30241984303a","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:12.021994Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:d91eacab4ac279a1be7de752db31479559a74fb93264df13217d3dc8a54cb237","observation_id":"868a9a4d-2155-48ea-b00e-50bbf09f2a3e","resolution":{"observed_at":"2026-08-15T23:08:12.189493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:12.161912Z","title":"type\": \"function","venue":null,"work_id":"0ba10be6-9603-4a45-87a9-e2f9a73b7d64","year":null},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:12.028546Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:c755933bb701821a3f97d1b6f1708b8933fb6a8a736bebdb9eab6ca605eff340","observation_id":"56d952e2-6119-4edb-ae33-2ee7fff44b3a","resolution":{"observed_at":"2026-08-15T23:08:12.168848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:08:13.462754Z","title":"In Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), pages 2877–2888, Online","venue":null,"work_id":"4845775d-5880-4a91-bfd7-5ae3f893ba26","year":2020},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.519127Z"},"links":{"citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:fcbfea50be54e00b898b3d9c452032e701b67099f03c43d458f621d74060973d","observation_id":"07fb300f-a7b1-4fdf-90e9-ea9af9cb7d51","resolution":{"observed_at":"2026-08-15T23:08:13.471681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.13233","last_updated":"2023-09-23T02:04:57Z","snapshot_observed_at":"2026-08-16T14:58:17.709886Z","submitted_at":"2023-09-23T02:04:57Z","title":"User Simulation with Large Language Models for Evaluating Task-Oriented Dialogue","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.13233","snapshot_observed_at":"2026-08-15T23:08:11.511635Z","title":"CoRR, abs/2309.13233","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.511635Z"},"links":{"cited_paper":"/paper/2309.13233","citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:df79995c5f8bf113a8d3b37aaa5b615428f411fe75a55d228d2b9bd457fe8eb6","observation_id":"c983d5c9-9f91-44b9-956b-035bc07ceb5c","resolution":{"observed_at":"2026-08-15T23:08:11.511635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-15T23:08:11.525565Z","title":"Preprint, arXiv:2407.21783","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.525565Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:8f3a634f574e8b26a6dc17baf7ef8764d00569407eafe80a8ae2d4ff9f7c26c6","observation_id":"a874f7cc-db01-4b80-9b9a-23b973640b92","resolution":{"observed_at":"2026-08-15T23:08:11.525565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.03706","last_updated":"2021-07-07T20:33:02Z","snapshot_observed_at":"2026-08-19T09:34:36.721994Z","submitted_at":"2021-06-07T15:17:03Z","title":"A Comprehensive Assessment of Dialog Evaluation Metrics","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.03706","snapshot_observed_at":"2026-08-15T23:08:11.541632Z","title":"Yi-Ting Yeh, Maxine Eskénazi, and Shikib Mehri","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","version":2},"reference_index":2763,"source":"pdf_text","source_observed_at":"2026-08-15T23:08:11.541632Z"},"links":{"cited_paper":"/paper/2106.03706","citing_paper":"/paper/2505.05445"},"observation_digest":"sha256:2267619384b790b5220e6ada2f22ec8de1782dc45943a7b19ae9c8e0b65e6d12","observation_id":"74478500-8350-4ab8-af79-28af5d0e775e","resolution":{"observed_at":"2026-08-15T23:08:11.541632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.05445","last_updated":"2025-07-21T12:42:09Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-19T22:10:13.008117Z","submitted_at":"2025-05-08T17:36:36Z","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations"},"reference_resolution":{"displayed":66,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":40,"verified_exact":0,"verified_fuzzy":25},"total_outbound_references":66},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 66 of 66 outbound references and 0 inbound Pith citation observations for arXiv:2505.05445."}