{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:E43PIF2SKPUBL46E3SQT57EHJ4","short_pith_number":"pith:E43PIF2S","schema_version":"1.0","canonical_sha256":"2736f4175253e815f3c4dca13efc874f0f1fe25673c6bfd67289ad4d892b2d59","source":{"kind":"arxiv","id":"2608.07770","version":1},"attestation_state":"computed","paper":{"title":"From Benchmark Performance to Tool Deployment: Human-in-the-Loop Anomaly Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Agit Yesiloz, Alexander Ur, Benjamin Wynn, Christopher Stokes, CJ George, Clint Kallenbach, Dakota Fulp, Gavin Smithson, Mark Swartz, Mike Szklarzewski, Nathan Debardeleben, Sharmistha Chakrabarti, William M. Jones","submitted_at":"2026-08-07T21:31:20Z","abstract_excerpt":"Automated anomaly detection methods often report strong performance on curated academic benchmarks, but their behavior under real-world industrial conditions is less clear. In this work, we evaluate 19 unsupervised anomaly detection models on the BowTie dataset, a challenging manufacturing dataset with reflective surfaces, subtle defects, and profile-specific variation. In contrast to benchmark results, we observe that model performance is less stable than typically reported on standard benchmarks such as MVTec AD, highly sensitive to preprocessing, and inconsistent across conditions, with no "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.07770","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-08-07T21:31:20Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"11dcb11fe6d56623ede4f280925a1639d714a2c1b1ab8211c7797f5d58d1f11d","abstract_canon_sha256":"54d38f00e077afa5473b9c75b4ea5214c4f6219af0274c5a9f92b64952ee5d15"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-11T00:15:55.588210Z","signature_b64":"BPp0fRj53pu1R9+bRwyS6dBT0zg9I0TJoFDmA1wSiWiqChXkoGNfCBFx/LZjiTZ9z32fYhNDPyw3F9p5N5U2BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2736f4175253e815f3c4dca13efc874f0f1fe25673c6bfd67289ad4d892b2d59","last_reissued_at":"2026-08-11T00:15:55.586044Z","signature_status":"signed_v1","first_computed_at":"2026-08-11T00:15:55.586044Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Benchmark Performance to Tool Deployment: Human-in-the-Loop Anomaly Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Agit Yesiloz, Alexander Ur, Benjamin Wynn, Christopher Stokes, CJ George, Clint Kallenbach, Dakota Fulp, Gavin Smithson, Mark Swartz, Mike Szklarzewski, Nathan Debardeleben, Sharmistha Chakrabarti, William M. Jones","submitted_at":"2026-08-07T21:31:20Z","abstract_excerpt":"Automated anomaly detection methods often report strong performance on curated academic benchmarks, but their behavior under real-world industrial conditions is less clear. In this work, we evaluate 19 unsupervised anomaly detection models on the BowTie dataset, a challenging manufacturing dataset with reflective surfaces, subtle defects, and profile-specific variation. In contrast to benchmark results, we observe that model performance is less stable than typically reported on standard benchmarks such as MVTec AD, highly sensitive to preprocessing, and inconsistent across conditions, with no "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.07770","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.07770/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.07770","created_at":"2026-08-11T00:15:55.586416+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.07770v1","created_at":"2026-08-11T00:15:55.586416+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.07770","created_at":"2026-08-11T00:15:55.586416+00:00"},{"alias_kind":"pith_short_12","alias_value":"E43PIF2SKPUB","created_at":"2026-08-11T00:15:55.586416+00:00"},{"alias_kind":"pith_short_16","alias_value":"E43PIF2SKPUBL46E","created_at":"2026-08-11T00:15:55.586416+00:00"},{"alias_kind":"pith_short_8","alias_value":"E43PIF2S","created_at":"2026-08-11T00:15:55.586416+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E43PIF2SKPUBL46E3SQT57EHJ4","json":"https://pith.science/pith/E43PIF2SKPUBL46E3SQT57EHJ4.json","graph_json":"https://pith.science/api/pith-number/E43PIF2SKPUBL46E3SQT57EHJ4/graph.json","events_json":"https://pith.science/api/pith-number/E43PIF2SKPUBL46E3SQT57EHJ4/events.json","paper":"https://pith.science/paper/E43PIF2S"},"agent_actions":{"view_html":"https://pith.science/pith/E43PIF2SKPUBL46E3SQT57EHJ4","download_json":"https://pith.science/pith/E43PIF2SKPUBL46E3SQT57EHJ4.json","view_paper":"https://pith.science/paper/E43PIF2S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.07770&json=true","fetch_graph":"https://pith.science/api/pith-number/E43PIF2SKPUBL46E3SQT57EHJ4/graph.json","fetch_events":"https://pith.science/api/pith-number/E43PIF2SKPUBL46E3SQT57EHJ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E43PIF2SKPUBL46E3SQT57EHJ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E43PIF2SKPUBL46E3SQT57EHJ4/action/storage_attestation","attest_author":"https://pith.science/pith/E43PIF2SKPUBL46E3SQT57EHJ4/action/author_attestation","sign_citation":"https://pith.science/pith/E43PIF2SKPUBL46E3SQT57EHJ4/action/citation_signature","submit_replication":"https://pith.science/pith/E43PIF2SKPUBL46E3SQT57EHJ4/action/replication_record"}},"created_at":"2026-08-11T00:15:55.586416+00:00","updated_at":"2026-08-11T00:15:55.586416+00:00"}