{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:XJONYQJTAG2JFZRG4IRUWHK33J","short_pith_number":"pith:XJONYQJT","schema_version":"1.0","canonical_sha256":"ba5cdc413301b492e626e2234b1d5bda666c66961b2b4ca4a10c86af64781ab3","source":{"kind":"arxiv","id":"2007.11154","version":2},"attestation_state":"computed","paper":{"title":"Rethinking CNN Models for Audio Classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Angela Yao, Dipika Singhania, Kamalesh Palanisamy","submitted_at":"2020-07-22T01:31:44Z","abstract_excerpt":"In this paper, we show that ImageNet-Pretrained standard deep CNN models can be used as strong baseline networks for audio classification. Even though there is a significant difference between audio Spectrogram and standard ImageNet image samples, transfer learning assumptions still hold firmly. To understand what enables the ImageNet pretrained models to learn useful audio representations, we systematically study how much of pretrained weights is useful for learning spectrograms. We show (1) that for a given standard model using pretrained weights is better than using randomly initialized wei"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2007.11154","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-07-22T01:31:44Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"8a1acbb4393308ba2ca9e44e4a5306a6af577d341461627547f5b4d966accbda","abstract_canon_sha256":"6a23fe2b0c4cf5c0b77e1ead6dd724ca955e29268ca2a1088da6a441e496675d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:51:28.190903Z","signature_b64":"2yz3k85koTS55nL9vIvVz+ICSEvRLk+rIfRVQAaPjiUiFf5AAd2/8BkWpoCK+y+2w3pRuK2+InQEN1pybb73BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba5cdc413301b492e626e2234b1d5bda666c66961b2b4ca4a10c86af64781ab3","last_reissued_at":"2026-07-05T01:51:28.190554Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:51:28.190554Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rethinking CNN Models for Audio Classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Angela Yao, Dipika Singhania, Kamalesh Palanisamy","submitted_at":"2020-07-22T01:31:44Z","abstract_excerpt":"In this paper, we show that ImageNet-Pretrained standard deep CNN models can be used as strong baseline networks for audio classification. Even though there is a significant difference between audio Spectrogram and standard ImageNet image samples, transfer learning assumptions still hold firmly. To understand what enables the ImageNet pretrained models to learn useful audio representations, we systematically study how much of pretrained weights is useful for learning spectrograms. We show (1) that for a given standard model using pretrained weights is better than using randomly initialized wei"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.11154","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2007.11154/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2007.11154","created_at":"2026-07-05T01:51:28.190609+00:00"},{"alias_kind":"arxiv_version","alias_value":"2007.11154v2","created_at":"2026-07-05T01:51:28.190609+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.11154","created_at":"2026-07-05T01:51:28.190609+00:00"},{"alias_kind":"pith_short_12","alias_value":"XJONYQJTAG2J","created_at":"2026-07-05T01:51:28.190609+00:00"},{"alias_kind":"pith_short_16","alias_value":"XJONYQJTAG2JFZRG","created_at":"2026-07-05T01:51:28.190609+00:00"},{"alias_kind":"pith_short_8","alias_value":"XJONYQJT","created_at":"2026-07-05T01:51:28.190609+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01743","citing_title":"InterCMDM: Block-Causal Diffusion for Autoregressive Human Interaction Generation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2310.01852","citing_title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","ref_index":68,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XJONYQJTAG2JFZRG4IRUWHK33J","json":"https://pith.science/pith/XJONYQJTAG2JFZRG4IRUWHK33J.json","graph_json":"https://pith.science/api/pith-number/XJONYQJTAG2JFZRG4IRUWHK33J/graph.json","events_json":"https://pith.science/api/pith-number/XJONYQJTAG2JFZRG4IRUWHK33J/events.json","paper":"https://pith.science/paper/XJONYQJT"},"agent_actions":{"view_html":"https://pith.science/pith/XJONYQJTAG2JFZRG4IRUWHK33J","download_json":"https://pith.science/pith/XJONYQJTAG2JFZRG4IRUWHK33J.json","view_paper":"https://pith.science/paper/XJONYQJT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2007.11154&json=true","fetch_graph":"https://pith.science/api/pith-number/XJONYQJTAG2JFZRG4IRUWHK33J/graph.json","fetch_events":"https://pith.science/api/pith-number/XJONYQJTAG2JFZRG4IRUWHK33J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XJONYQJTAG2JFZRG4IRUWHK33J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XJONYQJTAG2JFZRG4IRUWHK33J/action/storage_attestation","attest_author":"https://pith.science/pith/XJONYQJTAG2JFZRG4IRUWHK33J/action/author_attestation","sign_citation":"https://pith.science/pith/XJONYQJTAG2JFZRG4IRUWHK33J/action/citation_signature","submit_replication":"https://pith.science/pith/XJONYQJTAG2JFZRG4IRUWHK33J/action/replication_record"}},"created_at":"2026-07-05T01:51:28.190609+00:00","updated_at":"2026-07-05T01:51:28.190609+00:00"}