{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:G34VJ5IMQ7LITRMXOZD67GRLGC","short_pith_number":"pith:G34VJ5IM","schema_version":"1.0","canonical_sha256":"36f954f50c87d689c5977647ef9a2b309090f8eb92df73fb48b422612540f355","source":{"kind":"arxiv","id":"2502.20857","version":1},"attestation_state":"computed","paper":{"title":"JiTTER: Jigsaw Temporal Transformer for Event Reconstruction for Self-Supervised Sound Event Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Hyeonuk Nam, Yong-Hwa Park","submitted_at":"2025-02-28T08:55:20Z","abstract_excerpt":"Sound event detection (SED) has significantly benefited from self-supervised learning (SSL) approaches, particularly masked audio transformer for SED (MAT-SED), which leverages masked block prediction to reconstruct missing audio segments. However, while effective in capturing global dependencies, masked block prediction disrupts transient sound events and lacks explicit enforcement of temporal order, making it less suitable for fine-grained event boundary detection. To address these limitations, we propose JiTTER (Jigsaw Temporal Transformer for Event Reconstruction), an SSL framework designe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.20857","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2025-02-28T08:55:20Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"69c9d5a1fa209c512f15b8b9fcd8fef7d29e0462fc9f3c7a8794259daffe71a8","abstract_canon_sha256":"f91b4ec531880623f29059573153e6817ad3f8da4cc93a966a29e5b8c45d5814"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:41.191220Z","signature_b64":"GW0cGLWpTpqr9aTQ7Lznx78FyCOBd25yb4NUKhUnWIgiX3tsle77fxv20zbxNk571t29bfFrpatuTJgIrJqcBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36f954f50c87d689c5977647ef9a2b309090f8eb92df73fb48b422612540f355","last_reissued_at":"2026-07-05T10:21:41.190753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:41.190753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"JiTTER: Jigsaw Temporal Transformer for Event Reconstruction for Self-Supervised Sound Event Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Hyeonuk Nam, Yong-Hwa Park","submitted_at":"2025-02-28T08:55:20Z","abstract_excerpt":"Sound event detection (SED) has significantly benefited from self-supervised learning (SSL) approaches, particularly masked audio transformer for SED (MAT-SED), which leverages masked block prediction to reconstruct missing audio segments. However, while effective in capturing global dependencies, masked block prediction disrupts transient sound events and lacks explicit enforcement of temporal order, making it less suitable for fine-grained event boundary detection. To address these limitations, we propose JiTTER (Jigsaw Temporal Transformer for Event Reconstruction), an SSL framework designe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.20857","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.20857/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.20857","created_at":"2026-07-05T10:21:41.190812+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.20857v1","created_at":"2026-07-05T10:21:41.190812+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.20857","created_at":"2026-07-05T10:21:41.190812+00:00"},{"alias_kind":"pith_short_12","alias_value":"G34VJ5IMQ7LI","created_at":"2026-07-05T10:21:41.190812+00:00"},{"alias_kind":"pith_short_16","alias_value":"G34VJ5IMQ7LITRMX","created_at":"2026-07-05T10:21:41.190812+00:00"},{"alias_kind":"pith_short_8","alias_value":"G34VJ5IM","created_at":"2026-07-05T10:21:41.190812+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.07829","citing_title":"Auditory Intelligence: Understanding the World Through Sound","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G34VJ5IMQ7LITRMXOZD67GRLGC","json":"https://pith.science/pith/G34VJ5IMQ7LITRMXOZD67GRLGC.json","graph_json":"https://pith.science/api/pith-number/G34VJ5IMQ7LITRMXOZD67GRLGC/graph.json","events_json":"https://pith.science/api/pith-number/G34VJ5IMQ7LITRMXOZD67GRLGC/events.json","paper":"https://pith.science/paper/G34VJ5IM"},"agent_actions":{"view_html":"https://pith.science/pith/G34VJ5IMQ7LITRMXOZD67GRLGC","download_json":"https://pith.science/pith/G34VJ5IMQ7LITRMXOZD67GRLGC.json","view_paper":"https://pith.science/paper/G34VJ5IM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.20857&json=true","fetch_graph":"https://pith.science/api/pith-number/G34VJ5IMQ7LITRMXOZD67GRLGC/graph.json","fetch_events":"https://pith.science/api/pith-number/G34VJ5IMQ7LITRMXOZD67GRLGC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G34VJ5IMQ7LITRMXOZD67GRLGC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G34VJ5IMQ7LITRMXOZD67GRLGC/action/storage_attestation","attest_author":"https://pith.science/pith/G34VJ5IMQ7LITRMXOZD67GRLGC/action/author_attestation","sign_citation":"https://pith.science/pith/G34VJ5IMQ7LITRMXOZD67GRLGC/action/citation_signature","submit_replication":"https://pith.science/pith/G34VJ5IMQ7LITRMXOZD67GRLGC/action/replication_record"}},"created_at":"2026-07-05T10:21:41.190812+00:00","updated_at":"2026-07-05T10:21:41.190812+00:00"}