{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CJSPAIOYK53QUACFOKDFAQD7GA","short_pith_number":"pith:CJSPAIOY","schema_version":"1.0","canonical_sha256":"1264f021d857770a0045728650407f302ad847f959ff127a414c869c413a3602","source":{"kind":"arxiv","id":"2506.03278","version":1},"attestation_state":"computed","paper":{"title":"FailureSensorIQ: A Multi-Choice QA Dataset for Understanding Sensor Relationships and Failure Modes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Christodoulos Constantinides, Claudio Guerrero, Dhaval Patel, Jayant Kalagnanam, Shuxin Lin, Sunil Dagajirao Patil","submitted_at":"2025-06-03T18:05:10Z","abstract_excerpt":"We introduce FailureSensorIQ, a novel Multi-Choice Question-Answering (MCQA) benchmarking system designed to assess the ability of Large Language Models (LLMs) to reason and understand complex, domain-specific scenarios in Industry 4.0. Unlike traditional QA benchmarks, our system focuses on multiple aspects of reasoning through failure modes, sensor data, and the relationships between them across various industrial assets. Through this work, we envision a paradigm shift where modeling decisions are not only data-driven using statistical tools like correlation analysis and significance tests, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.03278","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-03T18:05:10Z","cross_cats_sorted":[],"title_canon_sha256":"c9928b448c328d294fd374cf5c5fa453e8564afba0993dbc4ab0a6a79bcbcb11","abstract_canon_sha256":"e095e4d38d1e748043f5ad80f1dbd2bdeb8d25058b5018133f104a097964911b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:23.986618Z","signature_b64":"S01zFLwr3yi/WpoLHxqzxANQE20sFTrgYSYa7DE8bzCGwFtxiKmZ9CyxFDyAggYMKx3v1y3+da0Q0tVoDBv5Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1264f021d857770a0045728650407f302ad847f959ff127a414c869c413a3602","last_reissued_at":"2026-07-05T11:15:23.986137Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:23.986137Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FailureSensorIQ: A Multi-Choice QA Dataset for Understanding Sensor Relationships and Failure Modes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Christodoulos Constantinides, Claudio Guerrero, Dhaval Patel, Jayant Kalagnanam, Shuxin Lin, Sunil Dagajirao Patil","submitted_at":"2025-06-03T18:05:10Z","abstract_excerpt":"We introduce FailureSensorIQ, a novel Multi-Choice Question-Answering (MCQA) benchmarking system designed to assess the ability of Large Language Models (LLMs) to reason and understand complex, domain-specific scenarios in Industry 4.0. Unlike traditional QA benchmarks, our system focuses on multiple aspects of reasoning through failure modes, sensor data, and the relationships between them across various industrial assets. Through this work, we envision a paradigm shift where modeling decisions are not only data-driven using statistical tools like correlation analysis and significance tests, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03278","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.03278/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.03278","created_at":"2026-07-05T11:15:23.986195+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.03278v1","created_at":"2026-07-05T11:15:23.986195+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03278","created_at":"2026-07-05T11:15:23.986195+00:00"},{"alias_kind":"pith_short_12","alias_value":"CJSPAIOYK53Q","created_at":"2026-07-05T11:15:23.986195+00:00"},{"alias_kind":"pith_short_16","alias_value":"CJSPAIOYK53QUACF","created_at":"2026-07-05T11:15:23.986195+00:00"},{"alias_kind":"pith_short_8","alias_value":"CJSPAIOY","created_at":"2026-07-05T11:15:23.986195+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.08614","citing_title":"DiagnosticIQ: A Benchmark for LLM-Based Industrial Maintenance Action Recommendation from Symbolic Rules","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07413","citing_title":"FORGE: Fine-grained Multimodal Evaluation for Manufacturing Scenarios","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CJSPAIOYK53QUACFOKDFAQD7GA","json":"https://pith.science/pith/CJSPAIOYK53QUACFOKDFAQD7GA.json","graph_json":"https://pith.science/api/pith-number/CJSPAIOYK53QUACFOKDFAQD7GA/graph.json","events_json":"https://pith.science/api/pith-number/CJSPAIOYK53QUACFOKDFAQD7GA/events.json","paper":"https://pith.science/paper/CJSPAIOY"},"agent_actions":{"view_html":"https://pith.science/pith/CJSPAIOYK53QUACFOKDFAQD7GA","download_json":"https://pith.science/pith/CJSPAIOYK53QUACFOKDFAQD7GA.json","view_paper":"https://pith.science/paper/CJSPAIOY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.03278&json=true","fetch_graph":"https://pith.science/api/pith-number/CJSPAIOYK53QUACFOKDFAQD7GA/graph.json","fetch_events":"https://pith.science/api/pith-number/CJSPAIOYK53QUACFOKDFAQD7GA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CJSPAIOYK53QUACFOKDFAQD7GA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CJSPAIOYK53QUACFOKDFAQD7GA/action/storage_attestation","attest_author":"https://pith.science/pith/CJSPAIOYK53QUACFOKDFAQD7GA/action/author_attestation","sign_citation":"https://pith.science/pith/CJSPAIOYK53QUACFOKDFAQD7GA/action/citation_signature","submit_replication":"https://pith.science/pith/CJSPAIOYK53QUACFOKDFAQD7GA/action/replication_record"}},"created_at":"2026-07-05T11:15:23.986195+00:00","updated_at":"2026-07-05T11:15:23.986195+00:00"}