{"as_of":"2026-08-24T01:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e55303ece4370c40b4c04209328530fc90eefd11f94eb4c4d2e1ac9d8c9de34e","coverage":[{"denominator":78,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":78,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T23:09:21.501802Z","state":"measured"},{"denominator":78,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":78,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.21016/citation-record","integrity":"/paper/2412.21016/integrity","json":"/paper/2412.21016/citation-record.json","paper":"/paper/2412.21016"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.112333Z","title":"Llm-based multi-agent systems for software engineering: Literature review, vision and the road ahead,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.112333Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:f86eb3e6c35e86da3b40ff3b79f11928d9d6d42f44aa80e9195b46d14c0f36b4","observation_id":"33674812-ffab-4a95-976f-72c92510a7b6","resolution":{"observed_at":"2026-08-10T23:09:21.112333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.925727Z","title":"Fuzz4all: Universal fuzzing with large language models,","venue":null,"work_id":"1a055f5d-cb62-4580-99f1-541322af4096","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.117484Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:b57f43db555022cff31a373d8505178e4190d16f18f2616fc6d650cf56ae259b","observation_id":"5fc046a3-bdd2-4014-8b3e-1983ec2a78de","resolution":{"observed_at":"2026-08-10T23:09:22.930521Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.122598Z","title":"Promptrobust: Towards evaluating the robustness of large language models on adversarial prompts,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.122598Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:95f7c64c35bf693245e4faa8af9e9b1d7910b5a9d90d7d520921214ec46914ae","observation_id":"3275661c-c057-4a9b-a995-da7dae0e8e3c","resolution":{"observed_at":"2026-08-10T23:09:21.122598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.128619Z","title":"Sentiment analysis for software engineering: How far can we go?","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.128619Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:b2afc4c146e817e1e67f44aba6a945ac92bc35efbed09a01b5263db3ad73406b","observation_id":"9a76929f-de3e-4593-98c0-fe274ac3ceb8","resolution":{"observed_at":"2026-08-10T23:09:21.128619Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.902275Z","title":"Mttm: Metamorphic testing for textual content modera- tion software,","venue":null,"work_id":"6de61770-adde-4fbf-b9ea-8648089180ad","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.135286Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:f9ba53621fb0e3d1e70b37ec1d8b2890bb1a4716997632e960b745bacc944de6","observation_id":"6734b065-b4a1-4a6e-9210-d9360d842021","resolution":{"observed_at":"2026-08-10T23:09:22.906364Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.889028Z","title":"Unilog: Automatic logging via llm and in- context learning,","venue":null,"work_id":"abef608c-d0d3-493f-9f02-483eaef54226","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.140407Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:cd6f5866e405a0a58844b2e37e76e7b4c40061cd940e9e19cce3f56b334eecc3","observation_id":"d130626d-b2ee-462e-9257-46e59a1dc201","resolution":{"observed_at":"2026-08-10T23:09:22.893731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.145332Z","title":"Managing extreme ai risks amid rapid progress,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.145332Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:e39211b771099ea996f24629b5d3b50ac982d0ba79a1c83a096aaccaff943b67","observation_id":"79788340-7863-497e-873d-a19591d619e9","resolution":{"observed_at":"2026-08-10T23:09:21.145332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.874175Z","title":"Chatgpt incorrectness detection in software reviews,","venue":null,"work_id":"c7f7ad59-2987-4aa1-94df-6ccab261ca40","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.149682Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:30041d6b8030b7b6eebc08654abdc37f57e5f4ff92dc15d76074812968ff3950","observation_id":"3ef7f9ef-f075-4cd6-aa71-e3074c995016","resolution":{"observed_at":"2026-08-10T23:09:22.879241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.860553Z","title":"Development in times of hype: How freelancers explore generative ai?","venue":null,"work_id":"4d9c6704-a9f5-4608-aaa4-2edda89b5047","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.154156Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:10f2acb4e6b62b4eb68ae579b31954548f20f26a7b893c054586bba691af7697","observation_id":"7d33d601-7a80-4fd3-af4d-501597825469","resolution":{"observed_at":"2026-08-10T23:09:22.865692Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.846970Z","title":"How far are we? the triumphs and trials of generative ai in learning soft- ware engineering,","venue":null,"work_id":"4beb75c0-72e5-4540-a317-0b0bfa5012f6","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.161003Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:93c79fe35871abf86ab9cc306080412a7d2a2010da5197d6efdeb78778f330cc","observation_id":"837bd9ab-c39e-446e-981c-43f9c694c2be","resolution":{"observed_at":"2026-08-10T23:09:22.851908Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.834151Z","title":"Text classification via large language models,","venue":null,"work_id":"7dd116c3-088d-4e40-a46d-f096fabd96b5","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.167523Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:d6d5d691dfe9ee79c93ed9dfaf40cf37f488baf5d98a6b82e2eff4b079f06669","observation_id":"fd1bcdfb-db8b-4d35-8f79-31a2200d7232","resolution":{"observed_at":"2026-08-10T23:09:22.838328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.820447Z","title":"Chatgpt outperforms crowd workers for text-annotation tasks,","venue":null,"work_id":"d0935049-ba2f-4fa1-abf3-e29981e8f673","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.173570Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:e1ad03c0f4c9edc10177be0b31dc2790004ef5dd000f334f790d413d052bd144","observation_id":"64384c7c-b40e-418f-b1d4-c50c6e9032ae","resolution":{"observed_at":"2026-08-10T23:09:22.824801Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.807134Z","title":"Can llms replace manual annotation of software engineering artifacts?","venue":null,"work_id":"094bdc2d-0c62-47b7-89a5-d969263d9839","year":2025},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.178258Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:3cc1c2618e81574e7b006083703f1ddb4d48135bed2842d0fb0f05fb7e62180c","observation_id":"c055efff-a4c3-4010-baa4-bae9e7b02262","resolution":{"observed_at":"2026-08-10T23:09:22.811293Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.792903Z","title":"Assessing the robustness of llm-based nlp software via automated testing,","venue":null,"work_id":"5298d54e-28b7-4d1a-a4fc-3a516f86f76d","year":2025},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.183227Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:28a4a1dfe865dd73807ad23fc5c736b299ab3219b120518e491354d5bc767bb6","observation_id":"a84229fe-463e-4dbd-8c3b-ac72a8d2a595","resolution":{"observed_at":"2026-08-10T23:09:22.798225Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.779230Z","title":"Nanofuzz: A usable tool for automatic test generation,","venue":null,"work_id":"3654a697-abdc-42d9-9b4d-1c3798a3ef2b","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.187144Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:ab5481410776cf41a56864a1cb3e42c7c130145681c067b900b5bc011b35e8d9","observation_id":"95d88076-b034-4506-8d6a-c8693001592a","resolution":{"observed_at":"2026-08-10T23:09:22.784085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.765465Z","title":"Toward stealthy backdoor attacks against speech recognition via elements of sound,","venue":null,"work_id":"315dfe64-a890-4c20-8585-471191910fe3","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.191723Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:6883203ade4b6d1417dcb9fcd9ddd147100a5810fb3f7221a3d2ef419f093b73","observation_id":"c6cdac78-0161-4d62-8492-52b18cba3161","resolution":{"observed_at":"2026-08-10T23:09:22.770094Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.751726Z","title":"Multitest: Physical-aware object insertion for testing multi-sensor fusion perception systems,","venue":null,"work_id":"bf1b5855-ac8b-4dd6-89fb-d93c7488325d","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.195394Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:75e889c899d5e68fbcdb98c29dce12264c2d06ae202bea1a7c83874c7444638f","observation_id":"f0cc6a6b-bf70-4b63-abc5-598523c33f69","resolution":{"observed_at":"2026-08-10T23:09:22.756528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.737057Z","title":"Glue-x: Evaluating natural language understanding models from an out-of-distribution generalization perspective,","venue":null,"work_id":"acbe8f73-16cd-4b91-ab0d-7c3521ef3ad6","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.200738Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:a8f685c5db9295607be9e709d7b58a839c4ce4f6136437691ca07fc142994bb9","observation_id":"83743ad6-c082-46a2-97b9-f349c64aa8f3","resolution":{"observed_at":"2026-08-10T23:09:22.741857Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.723779Z","title":"Black box adversarial prompting for foundation models,","venue":null,"work_id":"b4b92a84-68e4-4915-bf27-02049514a59b","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.206056Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:bcb9a99c550133aca17d8d54d624ce5b4aa72892ca10e310b9750438a0d5dabf","observation_id":"7ffae674-a0e3-4f5f-9391-f2cc5313b7ee","resolution":{"observed_at":"2026-08-10T23:09:22.728528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.710114Z","title":"On the robustness of chatgpt: An adversarial and out-of-distribution perspective,","venue":null,"work_id":"b59b3972-1406-4a93-82ce-a2b418859df0","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.210579Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:03a0da4257ecbf94e00fe9d79b54ffdd491cdba6d596aee660a6e174da92a307","observation_id":"356e6372-d97a-48e8-8266-4f3741b08d8b","resolution":{"observed_at":"2026-08-10T23:09:22.714924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.215684Z","title":"Software testing with large language models: Survey, landscape, and vision,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.215684Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:d87280b7c1888c51808844b08f0a5965acd4d9548b77fc1ac181130d50907f26","observation_id":"418568d6-1722-4b50-8143-500f96e957ac","resolution":{"observed_at":"2026-08-10T23:09:21.215684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/3664812","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.546630Z","title":"Llmeffichecker:understanding and testing efficiency degradation of large language models,","venue":null,"work_id":"197a4445-3b5a-44aa-a9ae-c75a9a8f2e5a","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.220908Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:795b7d8ad6a8d146ca6cbddc354463191e61d11313113b2fb2e471ef2ae880e7","observation_id":"e3204ae7-2767-41c9-94ec-6f5d62b0024a","resolution":{"observed_at":"2026-08-10T23:09:21.554208Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.687670Z","title":"Deepatash: Focused test generation for deep learning systems,","venue":null,"work_id":"ecee5a3c-a5a9-454c-93ff-d694ba65c5e8","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.226102Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:b2a9e510b26592bfaac2f003307c60bbee29304bde9ebc241b4d26618c133e0f","observation_id":"1eda7d60-e6fd-4ff6-93e0-fafca9f7216c","resolution":{"observed_at":"2026-08-10T23:09:22.692771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.671941Z","title":"Atom: Automated black-box testing of multi-label image classification systems,","venue":null,"work_id":"e56a3485-a50c-48ab-b3c3-d939f7417c3b","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.230876Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:4616513f8eda0975c5cbc488f4c7ad55f953c6980dbf8dbee808ab52820c51c7","observation_id":"6703d7f1-45fe-4da1-ae5f-31e97dc19e4c","resolution":{"observed_at":"2026-08-10T23:09:22.676766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.658207Z","title":"Repairing failure-inducing inputs with input reflection,","venue":null,"work_id":"912907bf-25e8-4bc6-8c7f-7ba21e244c16","year":2022},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.237814Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:f1056e244ca3cedb81f228a7085c20e286ea1e9cfbec1399c4129cba738743c7","observation_id":"cd36557c-9ad9-42e6-8f46-5780b88d30de","resolution":{"observed_at":"2026-08-10T23:09:22.663064Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.242969Z","title":"Generating natural language adversarial examples through probability weighted word saliency,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.242969Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:09946cc02a46c78e65ea6be0733acafc870060c1ddcc162fcc206c3af13b07f2","observation_id":"7fe55874-cc23-415b-a653-00cf44943bac","resolution":{"observed_at":"2026-08-10T23:09:21.242969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.248451Z","title":"Good debt or bad debt: Detecting semantic orientations in economic texts,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.248451Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:202d5a67cc2f56d950b182f4bbc88df5a7fc837d5e7c5d874c4a83b62fe30bd9","observation_id":"02524678-74b1-4ee3-bf31-77690262da23","resolution":{"observed_at":"2026-08-10T23:09:21.248451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.625504Z","title":"Character-level convolutional networks for text classification,","venue":null,"work_id":"fbc7d446-384e-41e3-861d-29ed95619f5d","year":2015},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.253715Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:8a4d9e347e75bb1d75bfbd4b01cd8a84917c6044a18a2407080eaa0a0413d5bd","observation_id":"1a6fa3f3-8177-4a6f-a6e9-f141eec723ba","resolution":{"observed_at":"2026-08-10T23:09:22.629873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.612719Z","title":"Seeing stars: exploiting class relationships for sentiment categorization with respect to rating scales,","venue":null,"work_id":"ed0c4114-d688-4a51-bee2-83a3d576bd68","year":2005},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.258746Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:3ba7a411b63749e9074920a246fbc947a318a23a69fc72a1599d87a75139948f","observation_id":"23b35352-b359-4bb7-a925-b8d464360702","resolution":{"observed_at":"2026-08-10T23:09:22.617183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.598239Z","title":"Beam search: faster and monotonic,","venue":null,"work_id":"df6c5e0e-338e-4a33-a4ed-91b5c0079807","year":2022},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.263321Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:b32a551bd3908fd2b8caa9f744aa906eede3878f9715c151caf7c48d0d1be838","observation_id":"9db1dfea-966a-4440-934a-1f376c4fa517","resolution":{"observed_at":"2026-08-10T23:09:22.603298Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.584913Z","title":"Chatgpt- resistant screening instrument for identifying non-programmers,","venue":null,"work_id":"e07ce643-c0fe-479a-b517-ce084bd848cd","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.268488Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:4a39e76dff24a8c8533913d7bed9ecbb79e96ac901076cdc1ddfd4eb53f2f703","observation_id":"6839bccf-15d0-4655-b4dc-20d562b05e0c","resolution":{"observed_at":"2026-08-10T23:09:22.589874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.571581Z","title":"Uncovering the causes of emotions in software developer communication using zero-shot llms,","venue":null,"work_id":"f4cf34c6-778d-432c-a642-f7c304213ad2","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.273594Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:7970cc5d46c67a32d2836fc2bd525c48728b860e5352b6eb3d2cf184f9e1b892","observation_id":"1ec37e3c-5686-439f-859f-92f35dd2dc7f","resolution":{"observed_at":"2026-08-10T23:09:22.576917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.557526Z","title":"Can automated text classification improve content analysis of software project data?","venue":null,"work_id":"91a76d43-0c8c-425f-b90f-4262c42109ae","year":2013},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.278981Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:11b4283dba9b4fc2e3b9840429e3978dac7340c3ee8be62cc87fad5ecbf1f4c6","observation_id":"fdffdce0-926e-474c-a83d-34b258b5b19b","resolution":{"observed_at":"2026-08-10T23:09:22.562736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.544538Z","title":"Hqa-attack: toward high quality black-box hard-label adversarial attack on text,","venue":null,"work_id":"6abecea6-9cfb-4410-afb6-ef63f5560c25","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.283195Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:fc81200c1c58aae46cddf343f9a36f040347111477a1599aea9e109061a849d7","observation_id":"bf5a8710-661a-4804-9b04-dcfe1db9398f","resolution":{"observed_at":"2026-08-10T23:09:22.549179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.530246Z","title":"Limeattack: Local explainable method for textual hard-label adversarial attack,","venue":null,"work_id":"8cae0ffe-6af4-42c1-ac85-c51956d62714","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.287860Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:bcd022a438c86ca41164864bbba796134a70b0987f23c3a17583646fae6c00c1","observation_id":"770485b4-bd9c-41ac-a857-1b3976cbc250","resolution":{"observed_at":"2026-08-10T23:09:22.535128Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.518042Z","title":"Texthacker: Learning based hybrid local search algorithm for text hard-label adversarial attack,","venue":null,"work_id":"2a12094d-e4f3-4eaa-9bda-70ad9792d1b3","year":2022},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.292989Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:3547cd5fc4eb3cfae9bd61d8c7e9adf4a2a4dc038a548fd298909f9dbf721c58","observation_id":"bb9810b9-a9c9-407c-8463-5f5d28578be6","resolution":{"observed_at":"2026-08-10T23:09:22.521916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.505707Z","title":"Natural language adversarial defense through synonym encoding,","venue":null,"work_id":"5a6245ee-9d5e-434e-b1af-5900609b6d5d","year":2021},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.297170Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:d090323a59d97f188b713047743020887f03b31a84313fbc98aa2dfacab92b9f","observation_id":"53cd349d-7b15-4647-8aad-59cfa80feb9c","resolution":{"observed_at":"2026-08-10T23:09:22.509802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.492189Z","title":"Leap: Efficient and automated test method for nlp software,","venue":null,"work_id":"d3bbbaf3-8f77-48ba-839b-e2c9abfcf802","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.302028Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:882f8d3ec639070c691af2c979727434d377a4f73ff9b26ba4cf3315721cfb99","observation_id":"1aa2771a-3009-4c58-9d41-9ef9025fe74e","resolution":{"observed_at":"2026-08-10T23:09:22.496569Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.479305Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist (extended abstract),","venue":null,"work_id":"9df81730-0d3c-463e-9f38-1eddafe453b3","year":2021},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.307005Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:3824cc924336ea585bbc3d8c886a7886faacf37e1dda1c1b5f720a780f8d952d","observation_id":"83b1b76b-6787-41e6-aa05-60c5b669d082","resolution":{"observed_at":"2026-08-10T23:09:22.483626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.464571Z","title":"Understanding the value of software engineering technologies,","venue":null,"work_id":"c1a306d6-1013-4bd6-ac86-06a5bbc49a78","year":2009},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.311384Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:efbb53f7d0ee04843ef0334f41920b03833140c8dcce73d4e7fba755f28e90dc","observation_id":"a4526533-c323-41d0-aeb1-765adb7d6204","resolution":{"observed_at":"2026-08-10T23:09:22.470028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.451940Z","title":"Enhancing speaker diarization with large language models: A contextual beam search approach,","venue":null,"work_id":"cb7a7d3e-4c65-43b8-b47b-469dccb73c18","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.315753Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:2104d3c02f9ff0fb372186f07bf926b2667346afe917dc002afdc2163abdf592","observation_id":"db34cc07-6585-4a53-bc40-589bf21f99ba","resolution":{"observed_at":"2026-08-10T23:09:22.456038Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.439374Z","title":"Conformal au- toregressive generation: Beam search with coverage guarantees,","venue":null,"work_id":"ffacfd74-6f21-4657-b3b1-67af10822902","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.321012Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:771e36ad2c6bfbd2be123fe33728a5153eeb77371725e04f6ad083a0557107b2","observation_id":"d54dc19c-591e-428f-9eb5-5f4ed2d8e374","resolution":{"observed_at":"2026-08-10T23:09:22.443625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.425544Z","title":"Hybrid filtered beam search algorithm for the optimization of monitoring patrols,","venue":null,"work_id":"b7cb3231-dbc5-4754-b99d-fedc3089b087","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.325846Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:5b79b70033bf615459fabd72f05507110c06387115136cf0fc6fc725f06e2c94","observation_id":"aef45b66-761c-4b47-b4a6-59e01dafb0df","resolution":{"observed_at":"2026-08-10T23:09:22.430249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.412526Z","title":"Large- scale language model rescoring on long-form data,","venue":null,"work_id":"e7980cd2-d72e-44a9-b181-eec815f9756c","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.331963Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:3d9cf6a201df79e1f1b0c6cd2e02533ebf09eab6a0617c830072533ec53a3c5c","observation_id":"94c062c3-9f67-4575-86b2-44ed789ac295","resolution":{"observed_at":"2026-08-10T23:09:22.416829Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.399067Z","title":"Beamqa: Multi-hop knowledge graph question answering with sequence-to-sequence prediction and beam search,","venue":null,"work_id":"5c7498b3-130b-4933-a067-199881f1d5c8","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.338073Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:06ba8f0bcf0cb294e38e2d84f664bdf6c2d281b67df2923cfd898a7b8c8288f8","observation_id":"4b0f0987-71b3-4720-8c34-8720c6a882d5","resolution":{"observed_at":"2026-08-10T23:09:22.403722Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.386532Z","title":"Iso/iec,","venue":null,"work_id":"38b5eb79-a3e2-460d-875f-a89377a78088","year":2017},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.344442Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:06023f97287a98f04fa59b7c462e0eed123d2483d1723f77e0fed2922b743e2f","observation_id":"2171988e-3c71-4cb6-92a8-9274de5bcff7","resolution":{"observed_at":"2026-08-10T23:09:22.390794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.373047Z","title":"Intriguing properties of neural networks,","venue":null,"work_id":"6181a484-fcc3-42a1-a4d5-cddc148e0311","year":2014},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.349539Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:67d27241f91523613108e5a68c905f525c9f040b98a1b1e25395d8b6a134d332","observation_id":"8e89aa7e-b2b6-4294-97dc-df6a2c1acaf5","resolution":{"observed_at":"2026-08-10T23:09:22.378054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.359267Z","title":"Textattack: A framework for adversarial attacks, data augmentation, and adversarial training in nlp,","venue":null,"work_id":"17d29262-0357-42b0-bb21-3c61b2ff52b1","year":2020},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.358164Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:2b7e4fbffb8172ce96eddade4186f0e38ccc11126c7ba9882df0f118acc6234b","observation_id":"9a19dcab-594a-416e-ae87-c45b2e6a2933","resolution":{"observed_at":"2026-08-10T23:09:22.364737Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.342991Z","title":"Fast adversarial attacks on language models in one GPU minute,","venue":null,"work_id":"79cc23c5-0148-46d0-a098-874e7a2c0339","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.362862Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:9194b1c49d7e2a59ff74ca1925ef46e87eeffa7a2173dbc17c20015cfd51e2f9","observation_id":"b548f926-232c-4463-8ccf-99137dfced67","resolution":{"observed_at":"2026-08-10T23:09:22.349577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.325736Z","title":"Hotflip: White-box adversarial examples for text classification,","venue":null,"work_id":"03aad210-cda0-4744-a2cd-d4a31bb2dee4","year":2018},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.367184Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:39e3421792525dc88599cbd049bf4af6a615570d16a3f56079dc56c504e1eb37","observation_id":"3456b88d-b2b2-461e-ac4e-bc8292936efa","resolution":{"observed_at":"2026-08-10T23:09:22.334432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.311741Z","title":"Towards improving adversarial training of nlp models,","venue":null,"work_id":"4af3a2c5-4a0c-40f7-b66f-cca68e0184c0","year":2021},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.371238Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:ceece977a97e7ce9f94d4e959ef276efade0bd3c07066756b7370e2f6404b24e","observation_id":"1cb91ef3-1283-438c-b482-d6067edec186","resolution":{"observed_at":"2026-08-10T23:09:22.316213Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.298822Z","title":"A recovering beam search algorithm for the one-machine dynamic total completion time scheduling problem,","venue":null,"work_id":"dbbe7905-498a-4bd4-9b1e-d44f3cf98489","year":2002},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.376016Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:ac4800918d07bd37bd7c98d171b90a5247a36e09cc38cc8ecdc6b7d31e65bd66","observation_id":"a4fc34d3-11d1-403c-a56b-dc14ae58942a","resolution":{"observed_at":"2026-08-10T23:09:22.303326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.284051Z","title":"Backtrack beam search for multiobjective scheduling problem,","venue":null,"work_id":"3b3aabed-0341-4c8f-ab24-431c6d566175","year":2003},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.380919Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:e86f54535a24a00e057affb7163bdd34f612a4d0d9c015076031daf6b20d38d0","observation_id":"3d2f3aa9-135a-4f03-91ea-e5a519594840","resolution":{"observed_at":"2026-08-10T23:09:22.289417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.270175Z","title":"Is bert really robust? a strong baseline for natural language attack on text classification and entailment,","venue":null,"work_id":"d8f60a0d-62bf-4706-ab25-0555a8034b8f","year":2020},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.386940Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:00c86a18ab1b07424d19e51973364b6f8cc55652f82eae4a030ab28e132b6be7","observation_id":"c4591ce7-dbcf-48ba-b0b5-40512ade96cd","resolution":{"observed_at":"2026-08-10T23:09:22.274372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.08220","last_updated":"2017-01-10T01:40:51Z","snapshot_observed_at":"2026-08-14T21:23:47.203647Z","submitted_at":"2016-12-24T21:36:07Z","title":"Understanding Neural Networks through Representation Erasure","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.08220","snapshot_observed_at":"2026-08-10T23:09:21.391726Z","title":"Understanding neural networks through representation erasure","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.391726Z"},"links":{"cited_paper":"/paper/1612.08220","citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:6e73c0ebda4fd1295b6ee7d7f1fb2cdd7cece5d5c9dfd42c9d0ff60be8058adb","observation_id":"9f2216c0-d92e-4a72-b0db-469daffdd1b9","resolution":{"observed_at":"2026-08-10T23:09:21.391726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.397121Z","title":"Wordnet: A lexical database for english,","venue":null,"work_id":null,"year":1995},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.397121Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:01469904449f9a241c5b21171431e4f4d478f17151545c0cdcc0a97fa696397b","observation_id":"c37a065d-974a-4ff8-a19e-5fab79d23ab7","resolution":{"observed_at":"2026-08-10T23:09:21.397121Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.256893Z","title":"Word-level textual adversarial attacking as combinatorial optimization,","venue":null,"work_id":"9bd79046-1c48-41ab-9b43-d1240f3ebb8f","year":2020},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.400795Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:b0cc93f3e62d497e7f1b53a6cffe730b4ac11a804113b2a190eadbd551a3642c","observation_id":"56b72cdf-06b7-4f4b-ae7a-3322a7ef246c","resolution":{"observed_at":"2026-08-10T23:09:22.261596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-08-17T20:30:34.016254Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-10T23:09:21.405749Z","title":"Mistral 7b,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.405749Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:58fae8c4feeade4be39213aeda032300e48884e29fbba5bd62835ffbebbca18b","observation_id":"16f620f4-327d-4154-b507-affe5b7aed6e","resolution":{"observed_at":"2026-08-10T23:09:21.405749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-10T23:09:21.411075Z","title":"Llama 2: Open foundation and fine-tuned chat models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.411075Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:d3faf3bdf944572eaa91914921126e5311d9bd3a76a3d85d11291ad27d7e6d40","observation_id":"64be3a36-ae2d-4a4c-b8a5-3dddb085b5d8","resolution":{"observed_at":"2026-08-10T23:09:21.411075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.243348Z","title":"Internlm2 technical report,","venue":null,"work_id":"cbf9a110-9b93-4d74-8d26-a7e4723ce878","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.415492Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:4742b7b36c00684802f16e3baf54895ebe9647838e39ade52d1dff8e2c526164","observation_id":"9d9c93eb-3fec-47a4-ba23-e26e00ff5fc9","resolution":{"observed_at":"2026-08-10T23:09:22.247527Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04652","last_updated":"2025-01-21T10:12:05Z","snapshot_observed_at":"2026-08-17T20:16:19.173618Z","submitted_at":"2024-03-07T16:52:49Z","title":"Yi: Open Foundation Models by 01.AI","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04652","snapshot_observed_at":"2026-08-10T23:09:21.419919Z","title":"Yi: Open foundation models by 01. ai,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.419919Z"},"links":{"cited_paper":"/paper/2403.04652","citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:12d25fabe03960d788abeb6583ec69848d6c5981b653eb2d7cf2367f096f3903","observation_id":"ba88e5d4-4337-467b-9f14-0d807046f222","resolution":{"observed_at":"2026-08-10T23:09:21.419919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.228911Z","title":"Stress test evaluation for natural language inference,","venue":null,"work_id":"0a48fe16-6f04-4fd7-a98c-c25c0a4398d4","year":2018},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.424990Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:9239c3c5564047ffeaa0ee1a6227187489d9ab63586c711f44e1544b58d1a774","observation_id":"6f53a992-0b25-4b12-a40b-0fee021bb096","resolution":{"observed_at":"2026-08-10T23:09:22.233591Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.213832Z","title":"Testing the limits: Unusual text inputs generation for mobile app crash detection with large language model,","venue":null,"work_id":"422676a6-badf-46bb-96e2-5dcb4479f407","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.428948Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:995703a98527ab421e025353d93c7da25584175ae75f61f9b339ba24164d7049","observation_id":"b502ec63-34fc-469d-838b-49a584c0565c","resolution":{"observed_at":"2026-08-10T23:09:22.218959Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.433205Z","title":"Textbugger: Generating adversarial text against real-world applications,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.433205Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:5135e91342c7021d241479d5624e30505e15c1b6f7ac5fe369ea0867385b1d08","observation_id":"70d22f26-bd95-44d1-be22-723f9ab587f2","resolution":{"observed_at":"2026-08-10T23:09:21.433205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11817","last_updated":"2025-02-13T08:11:25Z","snapshot_observed_at":"2026-08-16T19:44:00.811684Z","submitted_at":"2024-01-22T10:26:14Z","title":"Hallucination is Inevitable: An Innate Limitation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11817","snapshot_observed_at":"2026-08-10T23:09:21.436910Z","title":"Hallucination is inevitable: An innate limitation of large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.436910Z"},"links":{"cited_paper":"/paper/2401.11817","citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:1044d14a0bc0d46bc5d44c7fbbc16177e1fbedc2d409f74d044e9073bf2d2b1e","observation_id":"abe60e63-e933-42f6-a14e-1903770c4391","resolution":{"observed_at":"2026-08-10T23:09:21.436910Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16459","last_updated":"2023-09-28T14:09:58Z","snapshot_observed_at":"2026-08-18T11:22:59.236696Z","submitted_at":"2023-09-28T14:09:58Z","title":"Augmenting LLMs with Knowledge: A survey on hallucination prevention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16459","snapshot_observed_at":"2026-08-10T23:09:21.441235Z","title":"Augmenting llms with knowledge: A survey on hallucination prevention,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.441235Z"},"links":{"cited_paper":"/paper/2309.16459","citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:9795cdfcd4a88e9c02b9de3c01675760d04c0f380bec52842da48d420733ac16","observation_id":"e9bf3aa8-a061-45a8-bc64-1c5f44069dbd","resolution":{"observed_at":"2026-08-10T23:09:21.441235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.192269Z","title":"Coophance: Cooperative enhancement for robustness of deep learning systems,","venue":null,"work_id":"6d70c95b-e48d-4df6-b382-573f1140f05c","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.446430Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:996130f7d54b2911c5b7f23e7c64efd8910ba281181ea370e1582931948c6c0e","observation_id":"2eb9e727-e280-4086-93dd-3f36968b2241","resolution":{"observed_at":"2026-08-10T23:09:22.196783Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.178568Z","title":"Black-box testing of deep neural networks through test case diversity,","venue":null,"work_id":"9176ff8f-ce95-4e90-9d6b-770880e440d6","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.451171Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:29308924e657a2f264816d2dc75f1b79a397b3ee3bda4127d828ca5f5c91e011","observation_id":"326b48ef-266d-4ae1-b114-cff7df47625a","resolution":{"observed_at":"2026-08-10T23:09:22.184012Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.164733Z","title":"Dialtest: automated testing for recurrent- neural-network-driven dialogue systems,","venue":null,"work_id":"b0158891-5cb9-4e59-821a-dd7ea35bac06","year":2021},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.456009Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:c54ed1dde8a18e1b904fca9c6df1550e8140807f3510906049082d22e2563a6d","observation_id":"a852b0f6-d2d1-4547-9ea1-828af8349649","resolution":{"observed_at":"2026-08-10T23:09:22.169558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.460248Z","title":"Keeper: Automated testing and fixing of machine learning software,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.460248Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:d7e02cb9e892485b9428fad78b39d623986dad619eacff66a145237ba3a855db","observation_id":"3b9024e0-ef21-40ed-baff-0411a4256ad7","resolution":{"observed_at":"2026-08-10T23:09:21.460248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.151193Z","title":"Automated testing and improvement of named entity recognition systems,","venue":null,"work_id":"624cefee-96b9-433c-ba28-496f72cbc98a","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.464637Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:e0cabd4753bec3821d7ee733c19771f1d6147c3e9570180fdbc8a4435cb9d4ed","observation_id":"be0407e1-9bf2-4f16-88f5-666ea11fda5f","resolution":{"observed_at":"2026-08-10T23:09:22.156062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.137920Z","title":"Keeping llms aligned after fine-tuning: The crucial role of prompt templates,","venue":null,"work_id":"a19a9997-ae04-4976-be1a-57bbd80c5456","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.468984Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:cbbd6cff29567fa467a60e57db0aa5b1b97218362fcf933034f995ecb5306de0","observation_id":"7721979d-88a7-452c-b106-0bc9ea327962","resolution":{"observed_at":"2026-08-10T23:09:22.142585Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.125119Z","title":"Look before you leap: An exploratory study of uncertainty analysis for large language models,","venue":null,"work_id":"19e3725a-76b1-4555-a315-da2681565535","year":2025},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.474107Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:9af074f9c1cd72e86e791468a40f75fc5936b1774c7efe6b649cb9abe29db82f","observation_id":"a004e29f-f548-4591-95f4-3e135459b092","resolution":{"observed_at":"2026-08-10T23:09:22.129508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.482667Z","title":"Imperceptible content poisoning in llm-powered applications,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.482667Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:ab57eca32683a0ef71aa6e449ebe5fb4969f2e3d83d28cc52a9f927ed0fd4d44","observation_id":"753f0aea-8d64-4766-82db-6ce0c996a62e","resolution":{"observed_at":"2026-08-10T23:09:21.482667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.111403Z","title":"Revisiting out-of-distribution robustness in nlp: Bench- marks, analysis, and llms evaluations,","venue":null,"work_id":"3de24aca-660a-4507-af87-fec3d0ba74ca","year":2024},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.487088Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:ae697587efa7e8a0578e0bc77351b7be113008ea6c52aabca5e1dd0b2b606225","observation_id":"c9456c30-064f-4ae7-971b-6105118bb1b8","resolution":{"observed_at":"2026-08-10T23:09:22.116338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.097659Z","title":"Revisit input perturbation problems for llms: A unified robustness evaluation framework for noisy slot filling task,","venue":null,"work_id":"f51a5a12-c77c-40ee-8560-fe4e50e1f8a0","year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.492591Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:4e3a4c07a3f354b341ba26ac347b394d5a343c019ff2dd9b1d6b4885de742fe1","observation_id":"cf59aa33-b82f-4c3d-9b43-7b11c05dd111","resolution":{"observed_at":"2026-08-10T23:09:22.102241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:21.497519Z","title":"Ro- bustness over time: Understanding adversarial examples’ effectiveness 16 on longitudinal versions of large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.497519Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:bff3c76691575cf6dbc7c86e69f82a0051e34342288a85d39da09f4e40c89a1b","observation_id":"0ab261ff-e313-4c36-a2d3-e989f5cca9cd","resolution":{"observed_at":"2026-08-10T23:09:21.497519Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:09:22.083143Z","title":"degree in computer science and technology with the College of Computer Science and Software Engineering, Hohai University","venue":null,"work_id":"03c930b8-d98e-421e-8d7f-9aa835f5cc6e","year":2010},"citing_paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-10T23:09:21.501802Z"},"links":{"citing_paper":"/paper/2412.21016"},"observation_digest":"sha256:55cd93d7cb61799245b9ee75e36faa9090a5262667742b974b5930094b6b7735","observation_id":"6f7e4454-ddea-4549-97f0-3c53ed3ba49e","resolution":{"observed_at":"2026-08-10T23:09:22.088502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.21016","last_updated":"2025-03-17T13:42:06Z","latest_version":2,"primary_category":"cs.SE","snapshot_observed_at":"2026-08-14T23:33:27.849166Z","submitted_at":"2024-12-30T15:33:34Z","title":"Assessing the Robustness of LLM-based NLP Software via Automated Testing"},"reference_resolution":{"displayed":78,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":18,"verified_exact":1,"verified_fuzzy":59},"total_outbound_references":78},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 24 August 2026, this Paper Citation Record lists 78 of 78 outbound references and 0 inbound Pith citation observations for arXiv:2412.21016."}