{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:BKZSNTKGQB542E6EZAVQ7LTW7T","short_pith_number":"pith:BKZSNTKG","schema_version":"1.0","canonical_sha256":"0ab326cd46807bcd13c4c82b0fae76fce06b5a34473f76dd0034719ad64f3f98","source":{"kind":"arxiv","id":"2607.03201","version":1},"attestation_state":"computed","paper":{"title":"Deriving Benchmarking Datasets from Long-Form Recordings: Challenges and Opportunities","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Alejandrina Cristia, Alix Bourr\\'ee, Kaveri K. Sheth, Lawrence Borst, Loann Peurey, Marvin Lavechin, Okko R\\\"as\\\"anen, Sho Tsuji, Tarek Kunze","submitted_at":"2026-07-03T11:14:54Z","abstract_excerpt":"Long-form recordings (LFRs) of child-centered audio are ecologically valid sources for studying early language development, but three problems limit their use. First, LFR corpora are collected across sites with heterogeneous formats and consent structures, making cross-corpus use non-trivial. Second, without standardized benchmarks, assessing whether tools generalize across languages and conditions is hard. Third, ML workflows rarely respect privacy constraints governing sensitive child speech. This paper presents a framework addressing all three: a standardized collection of 27 child-centered"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.03201","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"eess.AS","submitted_at":"2026-07-03T11:14:54Z","cross_cats_sorted":["cs.LG","cs.SD"],"title_canon_sha256":"c911a5ced9b904b01451ba006a0f161e97d9278395518995a0e84cee9c5c984a","abstract_canon_sha256":"15b513d1a997dabbd03d291c62f96344241c885e6942b370f16be84d8bd70a06"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T01:16:46.423285Z","signature_b64":"Od6ak1l+f5ftWCGIrxREWVxUx3wpzvqfY9FXPtIaAy9NRMyo4dv92H724qRryU9x0DK4+CpFFOV1HFC6oG3kCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ab326cd46807bcd13c4c82b0fae76fce06b5a34473f76dd0034719ad64f3f98","last_reissued_at":"2026-07-07T01:16:46.422771Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T01:16:46.422771Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deriving Benchmarking Datasets from Long-Form Recordings: Challenges and Opportunities","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Alejandrina Cristia, Alix Bourr\\'ee, Kaveri K. Sheth, Lawrence Borst, Loann Peurey, Marvin Lavechin, Okko R\\\"as\\\"anen, Sho Tsuji, Tarek Kunze","submitted_at":"2026-07-03T11:14:54Z","abstract_excerpt":"Long-form recordings (LFRs) of child-centered audio are ecologically valid sources for studying early language development, but three problems limit their use. First, LFR corpora are collected across sites with heterogeneous formats and consent structures, making cross-corpus use non-trivial. Second, without standardized benchmarks, assessing whether tools generalize across languages and conditions is hard. Third, ML workflows rarely respect privacy constraints governing sensitive child speech. This paper presents a framework addressing all three: a standardized collection of 27 child-centered"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.03201","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.03201/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.03201","created_at":"2026-07-07T01:16:46.422838+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.03201v1","created_at":"2026-07-07T01:16:46.422838+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.03201","created_at":"2026-07-07T01:16:46.422838+00:00"},{"alias_kind":"pith_short_12","alias_value":"BKZSNTKGQB54","created_at":"2026-07-07T01:16:46.422838+00:00"},{"alias_kind":"pith_short_16","alias_value":"BKZSNTKGQB542E6E","created_at":"2026-07-07T01:16:46.422838+00:00"},{"alias_kind":"pith_short_8","alias_value":"BKZSNTKG","created_at":"2026-07-07T01:16:46.422838+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BKZSNTKGQB542E6EZAVQ7LTW7T","json":"https://pith.science/pith/BKZSNTKGQB542E6EZAVQ7LTW7T.json","graph_json":"https://pith.science/api/pith-number/BKZSNTKGQB542E6EZAVQ7LTW7T/graph.json","events_json":"https://pith.science/api/pith-number/BKZSNTKGQB542E6EZAVQ7LTW7T/events.json","paper":"https://pith.science/paper/BKZSNTKG"},"agent_actions":{"view_html":"https://pith.science/pith/BKZSNTKGQB542E6EZAVQ7LTW7T","download_json":"https://pith.science/pith/BKZSNTKGQB542E6EZAVQ7LTW7T.json","view_paper":"https://pith.science/paper/BKZSNTKG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.03201&json=true","fetch_graph":"https://pith.science/api/pith-number/BKZSNTKGQB542E6EZAVQ7LTW7T/graph.json","fetch_events":"https://pith.science/api/pith-number/BKZSNTKGQB542E6EZAVQ7LTW7T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BKZSNTKGQB542E6EZAVQ7LTW7T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BKZSNTKGQB542E6EZAVQ7LTW7T/action/storage_attestation","attest_author":"https://pith.science/pith/BKZSNTKGQB542E6EZAVQ7LTW7T/action/author_attestation","sign_citation":"https://pith.science/pith/BKZSNTKGQB542E6EZAVQ7LTW7T/action/citation_signature","submit_replication":"https://pith.science/pith/BKZSNTKGQB542E6EZAVQ7LTW7T/action/replication_record"}},"created_at":"2026-07-07T01:16:46.422838+00:00","updated_at":"2026-07-07T01:16:46.422838+00:00"}