{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JRAI5ACSW4HRVUHF7OBDFSXGRV","short_pith_number":"pith:JRAI5ACS","schema_version":"1.0","canonical_sha256":"4c408e8052b70f1ad0e5fb8232cae68d6b72cc8092f98c398f240a2316f61023","source":{"kind":"arxiv","id":"2505.18015","version":1},"attestation_state":"computed","paper":{"title":"SemSegBench & DetecBench: Benchmarking Reliability and Generalization Beyond Classification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"David Schader, Jonas Jakubassa, Margret Keuper, Mehmet Ege Ka\\c{c}ar, Nico Sharei, Ruben Weber, Shashank Agnihotri, Simon Kral","submitted_at":"2025-05-23T15:17:45Z","abstract_excerpt":"Reliability and generalization in deep learning are predominantly studied in the context of image classification. Yet, real-world applications in safety-critical domains involve a broader set of semantic tasks, such as semantic segmentation and object detection, which come with a diverse set of dedicated model architectures. To facilitate research towards robust model design in segmentation and detection, our primary objective is to provide benchmarking tools regarding robustness to distribution shifts and adversarial manipulations. We propose the benchmarking tools SEMSEGBENCH and DETECBENCH,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.18015","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-23T15:17:45Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"df37ff169539635bc76967ae3c4b7a83279dae4a5db9e7ad7052c0f1f02ca222","abstract_canon_sha256":"561829dd3940e5e2633dc7a3d5a708e82eb342ff5f466d1f7527df48f09c188c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:35.276750Z","signature_b64":"k3/uBgq8jhHahppOMXOK6cn87wUen/gKkpwSf5JEI+8s2LyvLhVZkrXzSzBk6A0QAVSPu+h3K5ZbrAoafnnTBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c408e8052b70f1ad0e5fb8232cae68d6b72cc8092f98c398f240a2316f61023","last_reissued_at":"2026-07-05T11:08:35.276197Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:35.276197Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SemSegBench & DetecBench: Benchmarking Reliability and Generalization Beyond Classification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"David Schader, Jonas Jakubassa, Margret Keuper, Mehmet Ege Ka\\c{c}ar, Nico Sharei, Ruben Weber, Shashank Agnihotri, Simon Kral","submitted_at":"2025-05-23T15:17:45Z","abstract_excerpt":"Reliability and generalization in deep learning are predominantly studied in the context of image classification. Yet, real-world applications in safety-critical domains involve a broader set of semantic tasks, such as semantic segmentation and object detection, which come with a diverse set of dedicated model architectures. To facilitate research towards robust model design in segmentation and detection, our primary objective is to provide benchmarking tools regarding robustness to distribution shifts and adversarial manipulations. We propose the benchmarking tools SEMSEGBENCH and DETECBENCH,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18015","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.18015/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.18015","created_at":"2026-07-05T11:08:35.276253+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.18015v1","created_at":"2026-07-05T11:08:35.276253+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18015","created_at":"2026-07-05T11:08:35.276253+00:00"},{"alias_kind":"pith_short_12","alias_value":"JRAI5ACSW4HR","created_at":"2026-07-05T11:08:35.276253+00:00"},{"alias_kind":"pith_short_16","alias_value":"JRAI5ACSW4HRVUHF","created_at":"2026-07-05T11:08:35.276253+00:00"},{"alias_kind":"pith_short_8","alias_value":"JRAI5ACS","created_at":"2026-07-05T11:08:35.276253+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JRAI5ACSW4HRVUHF7OBDFSXGRV","json":"https://pith.science/pith/JRAI5ACSW4HRVUHF7OBDFSXGRV.json","graph_json":"https://pith.science/api/pith-number/JRAI5ACSW4HRVUHF7OBDFSXGRV/graph.json","events_json":"https://pith.science/api/pith-number/JRAI5ACSW4HRVUHF7OBDFSXGRV/events.json","paper":"https://pith.science/paper/JRAI5ACS"},"agent_actions":{"view_html":"https://pith.science/pith/JRAI5ACSW4HRVUHF7OBDFSXGRV","download_json":"https://pith.science/pith/JRAI5ACSW4HRVUHF7OBDFSXGRV.json","view_paper":"https://pith.science/paper/JRAI5ACS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.18015&json=true","fetch_graph":"https://pith.science/api/pith-number/JRAI5ACSW4HRVUHF7OBDFSXGRV/graph.json","fetch_events":"https://pith.science/api/pith-number/JRAI5ACSW4HRVUHF7OBDFSXGRV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JRAI5ACSW4HRVUHF7OBDFSXGRV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JRAI5ACSW4HRVUHF7OBDFSXGRV/action/storage_attestation","attest_author":"https://pith.science/pith/JRAI5ACSW4HRVUHF7OBDFSXGRV/action/author_attestation","sign_citation":"https://pith.science/pith/JRAI5ACSW4HRVUHF7OBDFSXGRV/action/citation_signature","submit_replication":"https://pith.science/pith/JRAI5ACSW4HRVUHF7OBDFSXGRV/action/replication_record"}},"created_at":"2026-07-05T11:08:35.276253+00:00","updated_at":"2026-07-05T11:08:35.276253+00:00"}