{"as_of":"2026-08-04T09:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:600f3996ce53e333b8f67013e7c00f92d8766da7beb4546f6fed74c55760621e","coverage":[{"denominator":62,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":62,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-10T00:02:58.383840Z","state":"measured"},{"denominator":62,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":62,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.06873/citation-record","integrity":"/paper/2607.06873/integrity","json":"/paper/2607.06873/citation-record.json","paper":"/paper/2607.06873"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.387063Z","title":"ReAct: Synergizing reasoning and acting in language models","venue":null,"work_id":"dcede5c1-a91b-43d9-a097-8083603cb625","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:c6c0b339d7808379eb0a18aa42a9fbaee3269602039fad0fdaac9df9c1e7f7d6","observation_id":"8c9d1e7c-c348-422f-946b-62e9568cbc46","resolution":{"observed_at":"2026-07-10T00:06:38.388221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.388780Z","title":"Toolformer: Language models can teach themselves to use tools,","venue":null,"work_id":"d55164da-4150-4f21-99ae-664e11d9652a","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:a56739f7fe34b17120353140646d6b960bf74ca795a6a97c1d1d4e9ee2e34efd","observation_id":"72f10df0-056f-4523-8e0e-6aca88842e4b","resolution":{"observed_at":"2026-07-10T00:06:38.389906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-02T22:19:29.043854Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":"2406.12045","doi":"10.48550/arxiv.2406.12045","metadata_source":"pith","pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-07-11T01:37:42.734447Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","venue":"cs.AI","work_id":"6a8d8dc4-0cc0-4052-8109-abbcdcd4a962","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:450a7002da7895215f6fddb96f76c370f7e1953c2c7d0d1442b28049f5e92b01","observation_id":"4ea94b3d-f9b9-4888-9852-dd8a529e6290","resolution":{"observed_at":"2026-07-10T00:06:37.954013Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.392206Z","title":"Preventing repeated real world AI failures by cataloging incidents: The AI incident database,","venue":null,"work_id":"04f82ceb-1e89-4947-ab4f-907b61439719","year":2021},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:12aab4ea739856862b388201042bb3ea01dab38892ffd4a52f3ac95d2b91f452","observation_id":"4a4dfd04-19c2-4c43-a2fe-eafb3ce7c78b","resolution":{"observed_at":"2026-07-10T00:06:38.393408Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.351995Z","title":"RealHarm: A collection of real-world language model application failures,","venue":null,"work_id":"680a6106-96d1-47eb-8a5a-845608a9fe22","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:175cab7687ff5eae6a07c221ce4ce130b567b79394b762d8b257d8551f94d59d","observation_id":"4ffa99de-f735-4887-b6f9-e719cb61aae6","resolution":{"observed_at":"2026-07-10T00:06:38.353168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07982","last_updated":"2025-06-09T17:52:18Z","snapshot_observed_at":"2026-07-06T21:39:13.304260Z","submitted_at":"2025-06-09T17:52:18Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","version":1},"cited_work":{"arxiv_id":"2506.07982","doi":"10.48550/arxiv.2506.07982","metadata_source":"pith","pith_arxiv_id":"2506.07982","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","venue":"cs.AI","work_id":"3a498b1a-455f-4667-b572-c5216c99a89c","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"cited_paper":"/paper/2506.07982","citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:14594f3a4b48d5f5c90cc39b67e0f02247d6703ceeb76083146f1c18580d4452","observation_id":"f27b15c7-3df5-4d10-bf94-f07e62d6107d","resolution":{"observed_at":"2026-07-10T00:06:37.972449Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.129969+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.129969+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.383510Z","title":"WebArena: A realistic web environment for building autonomous agents,","venue":null,"work_id":"4496153c-9436-47b2-8cca-6c1643d00a18","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:c1bc7e3740afeb5d629066f488b6fb01342e3ce0ca181b7bf165e3b0c54e9925","observation_id":"e592fa41-3f52-480f-a16a-9c01898fae04","resolution":{"observed_at":"2026-07-10T00:06:38.384707Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.380080Z","title":"GAIA: A benchmark for general AI assistants,","venue":null,"work_id":"1f850405-8593-4d88-98f8-8f90ef2150ef","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:846906b421a044ece1ff3a9cd0807d67cd7099614328490acc1c824c58db1fd3","observation_id":"0f3b2126-34aa-4a65-826f-ade350a7d323","resolution":{"observed_at":"2026-07-10T00:06:38.381201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.376539Z","title":"AppWorld: A controllable world of apps and people for benchmarking interactive coding agents,","venue":null,"work_id":"8401e33f-476e-48e4-839d-a27859811428","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:b02082faab8c94816fc5ec7615e38fdc604e628d9dd8c24ac4265f12ad6c83c5","observation_id":"99a03855-90bb-4410-81dd-38744782f243","resolution":{"observed_at":"2026-07-10T00:06:38.377808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.399395Z","title":"SWE-bench: Can language models resolve real-world GitHub issues?","venue":null,"work_id":"67f71a17-451e-4c9d-8866-589d5e7e2244","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:7f838f57aeff9543e35ce49e1f84d6d62b584cf9e6088ae97c2cbdc970f5e8f8","observation_id":"c9221d46-79f0-4923-8502-f38c9021ec18","resolution":{"observed_at":"2026-07-10T00:06:38.401263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.397731Z","title":null,"venue":null,"work_id":"b0c27d99-18e0-48f6-a698-8ed6c098b86f","year":2011},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:d4abd03180a66624810e62b707d5f73a96406f69af804b1a928bb53cfecf6923","observation_id":"1f0725ff-4b9e-447f-b1c5-f27a6c1c8728","resolution":{"observed_at":"2026-07-10T00:06:38.398826Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.346356Z","title":"SpecOps: A fully automated AI agent testing framework in real-world GUI environments,","venue":null,"work_id":"85d8a534-c0b7-4823-b150-63893fe2f821","year":2026},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:2d1c8a18094fac8efa69dbac5a8ea5cc8bb1ae9291e5fd8707e56f6aa9c0157f","observation_id":"63c1db74-4e01-44c4-8c8c-8f522b8203c8","resolution":{"observed_at":"2026-07-10T00:06:38.347539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.00497","doi":"10.48550/arxiv.2601.00497","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"STELLAR: A search- based testing framework for large language model applications","venue":null,"work_id":"21d233de-26f8-492a-9344-6c57697c5195","year":2026},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:991410139a5f80b109e24f3b41744198f758fc27180e7d24aad961277a8814a3","observation_id":"daea19d1-e896-4fbd-9314-21ce8a47e832","resolution":{"observed_at":"2026-07-10T00:06:37.967164Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.419584Z","title":null,"venue":null,"work_id":"9cc11dfe-752e-46e4-a37b-d3872a4f2e60","year":2016},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:9f9d7fc26a7a1d3407e8f6a8e6d26a0ab328e4dace65b6c7796bafe4115c45b8","observation_id":"6143456a-537e-4313-8166-7a7f273e6ff7","resolution":{"observed_at":"2026-07-10T00:06:38.420670Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.421338Z","title":"A practitioner’s guide to process mining: Limitations of the directly-follows graph,","venue":null,"work_id":"2d8bfcd8-d517-433e-a59e-add1bdc570ef","year":2019},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:50310ed902f27889a367b30f9376f66c4c40dd61192b9db0af56add25998c8af","observation_id":"aec1170e-470f-4be9-8296-7486691ec429","resolution":{"observed_at":"2026-07-10T00:06:38.422481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.428153Z","title":"τ 3-bench: From text-only to multimodal, knowledge- aware agent evaluation,","venue":null,"work_id":"4ac8eb2f-cec1-42d8-b2cf-516f215d4046","year":2026},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:ca4af5697303270075bac2f912b3b7ed5291b5b1f0f109f21aa8bba467b88b25","observation_id":"57d39246-2796-4bff-b036-68890eadf922","resolution":{"observed_at":"2026-07-10T00:06:38.429430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.422960Z","title":"Event abstraction for process mining using supervised learning techniques,","venue":null,"work_id":"1414aa58-ab58-46bd-919e-d7df6f406848","year":2016},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:7ac882d56d9c2918e2d698a0a3b6cbcbceac368510407ec24757568ed286a535","observation_id":"e5e93c59-a7ef-4463-9592-d07d6a3f5697","resolution":{"observed_at":"2026-07-10T00:06:38.424048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.407607Z","title":"Event abstraction in process mining: Literature review and taxonomy,","venue":null,"work_id":"f5994e0a-62ea-4130-b1d0-64f4646a900a","year":2021},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:40a72cc56b64f2c80d2c4446438e5efa4b64f304ae52bad43ef6bdc834b0b55f","observation_id":"94bedfb8-7f23-49ae-8e03-092a82ad01ac","resolution":{"observed_at":"2026-07-10T00:06:38.408812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.412680Z","title":"Boundary value exploration for software analysis,","venue":null,"work_id":"737aa211-77aa-4e22-9b71-43d5d488df13","year":2020},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:e9251c02253f4f6f01ce4b48f17c8abb1714e1f0618ad9250e39c39af01cde87","observation_id":"a3e665c4-f53a-48ac-830b-15ce92ec3fb1","resolution":{"observed_at":"2026-07-10T00:06:38.413741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.426467Z","title":"Automated robustness testing of off-the-shelf software components,","venue":null,"work_id":"e8bc83a9-4dc1-4a81-86e1-389d20e5375f","year":1998},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:4b2cf45df2cd6cfe14ab3bbc82ce8f4708cfcfd5c7954c6c8a191b79de178699","observation_id":"94441f16-9357-421f-bc16-0c2ed032bfb8","resolution":{"observed_at":"2026-07-10T00:06:38.427615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.04370","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:37.974347Z","title":"τ- knowledge: Evaluating conversational agents over unstructured knowl- edge,","venue":null,"work_id":"0b88f354-50d7-46c9-bb81-2cce36a131b9","year":2026},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:11534a2c99cba890f6de9f03af239b42c6b2737d87664916b21ec878e2941cdd","observation_id":"954f9e5c-be91-4ff4-8818-3fd3c6f4b45d","resolution":{"observed_at":"2026-07-10T00:06:37.976247Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.390182Z","title":"ToolLLM: Facilitating large language models to master 16000+ real-world APIs,","venue":null,"work_id":"2d6e4ae2-a6b3-420b-9bb2-03aad7226356","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:46e1b1f8be8cb4d7a3e5dd0b986c615a4009ca1498c40eae1ff0f5a67c51bb8d","observation_id":"528726a0-4bbc-43f0-82da-02e1076c96fc","resolution":{"observed_at":"2026-07-10T00:06:38.391348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.00415","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:37.961746Z","title":"Towards self-evolving benchmarks: Synthesizing agent trajectories via test-time exploration under validate-by-reproduce paradigm","venue":null,"work_id":"447ad45f-cd0f-4901-8512-0c35542bde4a","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:d330866c5b170cb31665021640867df1609939a492a4f4bbb2fbff4d3ded1385","observation_id":"313d6d59-d163-401c-9181-18e859c0f448","resolution":{"observed_at":"2026-07-10T00:06:37.963929Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2410.11507","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:37.973618Z","title":"Revisiting benchmark and assessment: An agent-based exploratory dynamic evaluation framework for llms","venue":null,"work_id":"25f98d0b-1a74-494f-b6be-a9120415b2b8","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:00d5619006ec9e83367e01af0484bda7338ca59b2343c74931d2ec69930b5dae","observation_id":"97062318-6ecb-476d-8c13-da94c9bc37b9","resolution":{"observed_at":"2026-07-10T00:06:37.975111Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.00507","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:37.971374Z","title":"Graph2Eval: Automatic multimodal task generation for agents via knowledge graphs,","venue":null,"work_id":"6d3ef2d5-4034-405b-bcdd-f205d16da8a9","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:5ba95bfc6d84870f8c73c9bd8d3024033907ef0ea11e40d30c91edc9ed1e6dcb","observation_id":"86574250-54b4-41bf-a838-365082b4f56f","resolution":{"observed_at":"2026-07-10T00:06:37.973251Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-01T23:26:08.141299Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"cited_work":{"arxiv_id":"2505.17968","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.17968","snapshot_observed_at":"2026-07-10T00:06:37.955246Z","title":"arXiv preprint arXiv:2505.17968 , year=","venue":"cs.LG","work_id":"0d9a4447-37ad-49f8-99b3-19c44722d6c8","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"cited_paper":"/paper/2505.17968","citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:f07afc23ff28ba80f46c5b866e4e8b924b9ef6ff0b6daeee4f2ea3c0244fac54","observation_id":"5bddc95b-cf75-4d8b-bd2c-3374077b922a","resolution":{"observed_at":"2026-07-10T00:06:37.956842Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.371344Z","title":"Mining specifications,","venue":null,"work_id":"94c18af8-ec44-4c2c-b965-4f9e37831437","year":2002},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:3a6a58bea56a303abf9b5999ae5240d0492f17479cf17f794f19a76671f0babc","observation_id":"bdd38177-ce3c-4b83-9b07-36928981b4da","resolution":{"observed_at":"2026-07-10T00:06:38.372426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.393917Z","title":"Discovering models of software processes from event-based data,","venue":null,"work_id":"27132c23-9870-4008-854a-590d85298ea6","year":1998},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:eec1c26d210cb1e92275ae2b1b43555256b329040b0a96d26e8e4a8bf77d17e3","observation_id":"84bb6543-3b44-4012-a793-c6147a4fb1f0","resolution":{"observed_at":"2026-07-10T00:06:38.395137Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.401851Z","title":"Automatic generation of software behavioral models,","venue":null,"work_id":"d23b4b59-a8cc-467b-b1eb-64e0dcc0f1d7","year":2008},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:71fb393b6eeb51b9964381c533c39cd0ae81453f078640b42063a25982102c22","observation_id":"f89dc7ec-340b-4e20-b62f-18d00ca50c54","resolution":{"observed_at":"2026-07-10T00:06:38.403140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.409296Z","title":"Inferring models of concurrent systems from logs of their behavior with CSight,","venue":null,"work_id":"aadaf075-ff44-4bb0-b375-33e081c20de1","year":2014},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:b0ee6ad848342677e52f6dd7b09bb3416d7a0833f7cf63044e5e81727bc2375b","observation_id":"2dc6c576-bf93-4be9-bcc0-4dc77de44aad","resolution":{"observed_at":"2026-07-10T00:06:38.410430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.410917Z","title":"Workflow mining: Discovering process models from event logs,","venue":null,"work_id":"2770517a-2aeb-4f3e-9d0e-a337b517af4d","year":2004},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:10b22b84ae6c3b18013ebe8405468abda2acf7c9a676300150b489049ec030f3","observation_id":"a0c731b5-5311-457b-9da7-7e14474b33d1","resolution":{"observed_at":"2026-07-10T00:06:38.412159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.424592Z","title":"Discov- ering block-structured process models from event logs—a constructive approach,","venue":null,"work_id":"469031d6-dc71-46a9-890e-a8a42ba9871d","year":2013},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:88f32afa6ef1a152a513ee13c3efa6f5d78d12468a61c37d219028229ffc1c70","observation_id":"18aa5651-0695-48e9-822c-646501335467","resolution":{"observed_at":"2026-07-10T00:06:38.425884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.405837Z","title":"Applying graph reduction techniques for identifying structural conflicts in process models,","venue":null,"work_id":"07b30777-4717-4a60-8149-1d1ecf00f785","year":1999},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:afd8fe07a8df4b7dc1600c295432844cbf36112d19d1e357936c5ce7f625ff30","observation_id":"1f2e5127-a857-461b-abc0-1219399aa20c","resolution":{"observed_at":"2026-07-10T00:06:38.407049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.430274Z","title":"Learning regular sets from queries and counterexamples,","venue":null,"work_id":"f7272fe8-f42d-4fec-b465-11c9ab04ade2","year":1987},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:f51c5a4de4e37a0ce15bc4dc29bedd4cdfdfbf49a5d0ea4d4fc9b063082defe8","observation_id":"d81ef0ce-87de-4eb1-a2a2-1871b5a90b68","resolution":{"observed_at":"2026-07-10T00:06:38.431693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.414223Z","title":"Unsupervised dialog structure learning,","venue":null,"work_id":"ee07b3c5-8519-4d36-bf8b-6cd33ad7ad8e","year":2019},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:4abe2b3dba460886a6b233cfac2d3458b23b890b2cd70d6272198b6925896986","observation_id":"9e496994-d97f-432f-b6da-538d382c9579","resolution":{"observed_at":"2026-07-10T00:06:38.415418Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.391907Z","title":"Dialog2Flow: Pre-training soft-contrastive action-driven sentence embeddings for automatic dia- log flow extraction,","venue":null,"work_id":"13746e4c-101b-4b43-a6fe-49f17d2b4ed8","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:d088b3b08089cf7581e406125aecfd47ae6cd788b63acf7e3ef9ae2b358f08e9","observation_id":"8b21782c-e4f6-4cf1-8c76-f9e32d4497df","resolution":{"observed_at":"2026-07-10T00:06:38.393236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.403829Z","title":"Agenda-based user simulation for bootstrapping a POMDP dialogue system,","venue":null,"work_id":"31398511-6caa-4c5e-bb5d-16330d6159ae","year":2007},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:68f4903bf0068104afdf5169fe182b546b4af56bad567114e862c5961c0d3cef","observation_id":"47d02b58-9eb9-4c9d-8561-ca85fee5e5e5","resolution":{"observed_at":"2026-07-10T00:06:38.404999Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.405993Z","title":"ConvLab-2: An open-source toolkit for building, evaluating, and diagnosing dialogue systems,","venue":null,"work_id":"9f5a3aa7-3570-4098-8afe-25b387d95c5a","year":2020},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:fec715486c4c855e8977a21e1772fea91ad4624c46efd0a674e40e250396a587","observation_id":"8b43514d-7a55-415b-94e7-19673430fb4d","resolution":{"observed_at":"2026-07-10T00:06:38.407241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.384895Z","title":"A taxonomy of model- based testing approaches,","venue":null,"work_id":"54c4e6b2-2fbf-4590-87a1-07218c9cd816","year":2012},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:7651caf4f5d21c572bbd6bf1902e2dc32786bac06ccd7b5d295834cd7fbf0119","observation_id":"2fc55fc5-d849-4b04-aa23-93c0f960bcdb","resolution":{"observed_at":"2026-07-10T00:06:38.385993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.388516Z","title":"Principles and methods of testing finite state machines—a survey,","venue":null,"work_id":"73b5409a-ac05-46ca-b338-2463c67bb34c","year":1996},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:47536486338b40bbbd9583d0f192223537dc97d37cbbdd97485aaa39df90934a","observation_id":"75ce28d8-4ae0-40ca-ba6d-2e5f3c4d9ca9","resolution":{"observed_at":"2026-07-10T00:06:38.389665Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.395511Z","title":"Testing software design modeled by finite-state machines,","venue":null,"work_id":"83c208f5-69cf-482a-a3b5-cde6e4e39600","year":1978},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:ce2af597230ccc22cfdccf79c02367236d087752871a799832a648d99eb488c1","observation_id":"f73ca176-6c76-43e6-9d34-280c4bf33ab5","resolution":{"observed_at":"2026-07-10T00:06:38.396696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.383213Z","title":"RESTler: Stateful REST API fuzzing,","venue":null,"work_id":"5b793712-cb5e-41e5-a09f-1ac09f49ba0e","year":2019},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:86e18e5c8c4c5d1171eadb3f9790ac2b821996c5c9594b4ace1094a2b0e80581","observation_id":"25e0e7b8-c0d7-4dfd-bdc5-0355e81b625d","resolution":{"observed_at":"2026-07-10T00:06:38.384367Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.372714Z","title":"RESTful API automated test case generation with Evo- Master,","venue":null,"work_id":"c92a652e-731c-42a0-9221-eaa4a72ee2ae","year":2019},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:d2023d87625e48d96def7438e5eb0d2b3a8256a068ef0a19e30fadebde9c343a","observation_id":"413c8bf1-30d6-41dd-9201-7823baa165db","resolution":{"observed_at":"2026-07-10T00:06:38.373884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.349858Z","title":"Morest: Model-based RESTful API testing with execution feedback,","venue":null,"work_id":"5dce314c-5a54-4d89-9e5a-582cb97c2608","year":2022},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:bc47978c1e1d802a513dc8c8766634d7a3f2311964edd557eb17844897bf7ab2","observation_id":"9c71df55-6f0d-40e5-86cb-c486edd2c362","resolution":{"observed_at":"2026-07-10T00:06:38.351133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.374445Z","title":"KAT: Dependency-aware automated API testing with large language models,","venue":null,"work_id":"b2bd4ef3-dedc-4765-80e8-708ad6ef1753","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:0daa4af15c3e7183a30c9fb6a6d88e9ed102b76cf54ce67bd93a49ecf3326c0a","observation_id":"fdfc078e-4aac-477b-a422-f5bbcb06dd86","resolution":{"observed_at":"2026-07-10T00:06:38.375692Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.407771Z","title":"Testing RESTful APIs: A survey,","venue":null,"work_id":"fea21a97-42a0-4ae8-92b2-71721e7196ea","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:386e636bc49b866907a16e3e64a2a2ea0087c0ee1ee5750938ad6c7fbcb25066","observation_id":"5ccd9764-a37b-474b-b12d-7afdcb0079dd","resolution":{"observed_at":"2026-07-10T00:06:38.408976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.411221Z","title":"An empirical evaluation of using large language models for automated unit test generation,","venue":null,"work_id":"a22a948e-2c9d-4487-8f40-e073a1f67e91","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:0e627f7e14b8d792b07ec47565ba285b5648a4b1443a7b7b3297c22b1a013f97","observation_id":"2b3b289f-f2e8-4e78-98d5-f07c0cd671c2","resolution":{"observed_at":"2026-07-10T00:06:38.412345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.416353Z","title":"CoverUp: Effective high coverage test generation for Python,","venue":null,"work_id":"a5bf595d-c2b8-4f43-aea9-ea638da11615","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:a1607fe6f669b239b84ec91710b60208d7c000679a366890fce3b16f66e7aafc","observation_id":"5352081e-274f-4b47-bb01-3889218a77e5","resolution":{"observed_at":"2026-07-10T00:06:38.417481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.415993Z","title":"Evaluating and improving ChatGPT for unit test generation,","venue":null,"work_id":"6034b27e-e14c-4e73-90a4-75a4a58f8d21","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:de54bf2c51317e2a43ed4e8ea91e9eb6af7414724bc0d1fa4e079d2180bac607","observation_id":"a19faacc-39ec-42e5-9c38-01a65f71e8cd","resolution":{"observed_at":"2026-07-10T00:06:38.417180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.409474Z","title":"CodaMosa: Escaping coverage plateaus in test generation with pre-trained large language models,","venue":null,"work_id":"aae2f8f5-47e7-4723-ba40-ac50b3eafe98","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:a27d21dad7c615f01c683650923709d0d77441f4515c63e3b2dea28bc097c26d","observation_id":"c6508104-bd2e-4132-b3b5-e0026d11b7f7","resolution":{"observed_at":"2026-07-10T00:06:38.410662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03095","last_updated":"2025-03-31T13:13:27Z","snapshot_observed_at":"2026-07-06T18:57:22.433852Z","submitted_at":"2024-08-06T10:52:41Z","title":"TestART: Improving LLM-based Unit Testing via Co-evolution of Automated Generation and Repair Iteration","version":6},"cited_work":{"arxiv_id":"2408.03095","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.03095","snapshot_observed_at":"2026-07-10T00:06:37.968483Z","title":"Improving llm-based unit test generation via template-based repair","venue":"cs.SE","work_id":"14eda6da-904c-4e2a-af93-d226b0281ec6","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"cited_paper":"/paper/2408.03095","citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:299d7e47f53fe5c7955e994fc5e796ed5e12db5d8125b6fb5872be00234e4a55","observation_id":"74a8218e-fa88-4365-9149-1b2c1b93a272","resolution":{"observed_at":"2026-07-10T00:06:37.970137Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.361453Z","title":"The oracle problem in software testing: A survey","venue":null,"work_id":"fd9c3fd4-a9a0-4be7-8284-ff0a990a96f9","year":2015},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:7a59795a4aba2dbb561e0ec4c9f6fcb0ed620cf547038c72430e0cfd888d340e","observation_id":"d2ee0e6d-23e2-41d1-865d-8836ec6aaeef","resolution":{"observed_at":"2026-07-10T00:06:38.362723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.367375Z","title":"Pseudo-oracles for non-testable programs,","venue":null,"work_id":"d502e25a-8488-41d6-85a0-45321bf83977","year":1981},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:a2587a6de7666faf48569f4926b8839ffe21ed2d491eb72c4666f80abe06e1dc","observation_id":"91b24357-a3aa-40bb-a95e-781f191ad93b","resolution":{"observed_at":"2026-07-10T00:06:38.368635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.355218Z","title":"On testing non-testable programs,","venue":null,"work_id":"1c9fc5bb-0e20-4d93-8641-8c567d31df2c","year":1982},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:eded76b5a34bbf4be21008d1fff076bfc6cb36f4fb912c30f34b6de14c78772f","observation_id":"e755b481-bc90-4bad-9642-d31ac5d8b987","resolution":{"observed_at":"2026-07-10T00:06:38.356276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.377998Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":null,"work_id":"cccce1f9-7736-4f3d-8edd-e8144f4dd4b0","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:776e51d5846c44c667ba516c43d7c22b59dfe1c00429a0a96ed4b6aea895a20a","observation_id":"5280139e-b263-4a46-9432-063d6ccc5e10","resolution":{"observed_at":"2026-07-10T00:06:38.379148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.414605Z","title":"Large language models cannot self-correct reasoning yet","venue":null,"work_id":"8ec9480a-3243-490b-90b3-69edf7b7205a","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:aa320b487166dbde806b0f960d76dcee45270ecd6554ed44b9cec7b37ee08528","observation_id":"8919763f-e676-4082-b6e1-3691f34f621c","resolution":{"observed_at":"2026-07-10T00:06:38.415797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08115","last_updated":"2024-08-03T21:25:31Z","snapshot_observed_at":"2026-08-03T17:28:12.454636Z","submitted_at":"2024-02-12T23:11:01Z","title":"On the Self-Verification Limitations of Large Language Models on Reasoning and Planning Tasks","version":2},"cited_work":{"arxiv_id":"2402.08115","doi":"10.48550/arxiv.2402.08115","metadata_source":"pith","pith_arxiv_id":"2402.08115","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"On the self-verification limitations of large language models on reasoning and planning tasks","venue":"cs.AI","work_id":"0e539fa7-dc00-479f-bafa-8bda539f2f5d","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"cited_paper":"/paper/2402.08115","citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:61c36706047478854fbc416fe39ede2aa2494d47eac2e8ff89605b625a33ac70","observation_id":"56f5958e-caa8-4320-9086-ead614e1fd1c","resolution":{"observed_at":"2026-07-10T00:06:37.970034Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.370968Z","title":"Self-Refine: Iterative refinement with self-feedback,","venue":null,"work_id":"769df8fb-dbf6-4ea5-8b1f-028ef73398aa","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:9e8728434f24ba382c83471b1219f9eef9cad98e5abdf1df70044e420f56e447","observation_id":"fc6e1223-e3df-467e-90f8-89dcda18765c","resolution":{"observed_at":"2026-07-10T00:06:38.372182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.376195Z","title":"CRITIC: Large language models can self-correct with tool-interactive critiquing,","venue":null,"work_id":"e512635d-48a0-4eb7-8be0-ac15d2b8a660","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:ffed36c72cdf1275bc75a407380ee5a21c932a95ee6eb730e0b0e161418323d9","observation_id":"ffd8ae57-7bb3-41f1-a318-d4876be7b2df","resolution":{"observed_at":"2026-07-10T00:06:38.377391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.397556Z","title":"Metamorphic testing: A review of challenges and opportunities,","venue":null,"work_id":"fbb46226-564b-4d46-aaec-881b695f7b10","year":2018},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:9bb436c9a048c9c05f8d0670adeeb2be806733cb5280d51baebe81d87cf4d66d","observation_id":"4b153557-3da1-4ffe-b1b5-bd2cff717884","resolution":{"observed_at":"2026-07-10T00:06:38.398672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.401698Z","title":"Not what you’ve signed up for: Compromising real-world LLM- integrated applications with indirect prompt injection,","venue":null,"work_id":"4a1d50d5-4caf-46e9-86d9-f2fc58b5b7c0","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:8037808aa6a7e462066cd87b881e2bbd1f9970c75ef33a27e3c24806d39217a1","observation_id":"45b02c00-7cdd-4acc-b72b-106b5381a916","resolution":{"observed_at":"2026-07-10T00:06:38.403085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.417794Z","title":"Ignore previous prompt: Attack techniques for language models","venue":null,"work_id":"e1fb2557-e604-461b-8d64-d005e35c230b","year":2022},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:5a4f39fc37c0965aa1d5bd3b5147408b11b086d79a72f664795a7771f81f59e8","observation_id":"587a72f6-11a9-45de-b57e-fa05e453fc52","resolution":{"observed_at":"2026-07-10T00:06:38.419028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","latest_version":1,"primary_category":"cs.SE","snapshot_observed_at":"2026-08-02T02:41:53.958247Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents"},"reference_resolution":{"displayed":62,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":2,"verified_exact":9,"verified_fuzzy":50},"total_outbound_references":62},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 62 of 62 outbound references and 0 inbound Pith citation observations for arXiv:2607.06873."}