{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:O7B365WCB3EUNAHEWD732DSIEM","short_pith_number":"pith:O7B365WC","schema_version":"1.0","canonical_sha256":"77c3bf76c20ec94680e4b0ffbd0e482309d6992d4a8a7204a9c41ec8eaa26b08","source":{"kind":"arxiv","id":"2111.10882","version":1},"attestation_state":"computed","paper":{"title":"Geometry-Aware Multi-Task Learning for Binaural Audio Generation from Video","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Kristen Grauman, Rishabh Garg, Ruohan Gao","submitted_at":"2021-11-21T19:26:45Z","abstract_excerpt":"Binaural audio provides human listeners with an immersive spatial sound experience, but most existing videos lack binaural audio recordings. We propose an audio spatialization method that draws on visual information in videos to convert their monaural (single-channel) audio to binaural audio. Whereas existing approaches leverage visual features extracted directly from video frames, our approach explicitly disentangles the geometric cues present in the visual stream to guide the learning process. In particular, we develop a multi-task framework that learns geometry-aware features for binaural a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.10882","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-11-21T19:26:45Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"fbe8e10d448ab48bd40f9f05eff78696d01bd95fb0e58dbd4784c253fc9f16c4","abstract_canon_sha256":"ba066f3dbd4a9e55ca7f82ee56ae27c5c004c98e04ec87a58ce6f9d0ca4fdb91"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:33:57.860249Z","signature_b64":"P2AoN17nwae2x8JxHKqYbkOcBdNbwxANkIpadEtpgB0c4AuWNa92jPPuAoclnqQd2ZRjyGZ1Fxh7PDpENWZCDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77c3bf76c20ec94680e4b0ffbd0e482309d6992d4a8a7204a9c41ec8eaa26b08","last_reissued_at":"2026-07-05T03:33:57.859709Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:33:57.859709Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Geometry-Aware Multi-Task Learning for Binaural Audio Generation from Video","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Kristen Grauman, Rishabh Garg, Ruohan Gao","submitted_at":"2021-11-21T19:26:45Z","abstract_excerpt":"Binaural audio provides human listeners with an immersive spatial sound experience, but most existing videos lack binaural audio recordings. We propose an audio spatialization method that draws on visual information in videos to convert their monaural (single-channel) audio to binaural audio. Whereas existing approaches leverage visual features extracted directly from video frames, our approach explicitly disentangles the geometric cues present in the visual stream to guide the learning process. In particular, we develop a multi-task framework that learns geometry-aware features for binaural a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.10882","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.10882/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.10882","created_at":"2026-07-05T03:33:57.859767+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.10882v1","created_at":"2026-07-05T03:33:57.859767+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.10882","created_at":"2026-07-05T03:33:57.859767+00:00"},{"alias_kind":"pith_short_12","alias_value":"O7B365WCB3EU","created_at":"2026-07-05T03:33:57.859767+00:00"},{"alias_kind":"pith_short_16","alias_value":"O7B365WCB3EUNAHE","created_at":"2026-07-05T03:33:57.859767+00:00"},{"alias_kind":"pith_short_8","alias_value":"O7B365WC","created_at":"2026-07-05T03:33:57.859767+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30940","citing_title":"Towards Streaming Synchronized Spatial Audio Generation via Autoregressive Diffusion Transformer","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05731","citing_title":"FoleyDesigner: Immersive Stereo Foley Generation with Precise Spatio-Temporal Alignment for Film Clips","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O7B365WCB3EUNAHEWD732DSIEM","json":"https://pith.science/pith/O7B365WCB3EUNAHEWD732DSIEM.json","graph_json":"https://pith.science/api/pith-number/O7B365WCB3EUNAHEWD732DSIEM/graph.json","events_json":"https://pith.science/api/pith-number/O7B365WCB3EUNAHEWD732DSIEM/events.json","paper":"https://pith.science/paper/O7B365WC"},"agent_actions":{"view_html":"https://pith.science/pith/O7B365WCB3EUNAHEWD732DSIEM","download_json":"https://pith.science/pith/O7B365WCB3EUNAHEWD732DSIEM.json","view_paper":"https://pith.science/paper/O7B365WC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.10882&json=true","fetch_graph":"https://pith.science/api/pith-number/O7B365WCB3EUNAHEWD732DSIEM/graph.json","fetch_events":"https://pith.science/api/pith-number/O7B365WCB3EUNAHEWD732DSIEM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O7B365WCB3EUNAHEWD732DSIEM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O7B365WCB3EUNAHEWD732DSIEM/action/storage_attestation","attest_author":"https://pith.science/pith/O7B365WCB3EUNAHEWD732DSIEM/action/author_attestation","sign_citation":"https://pith.science/pith/O7B365WCB3EUNAHEWD732DSIEM/action/citation_signature","submit_replication":"https://pith.science/pith/O7B365WCB3EUNAHEWD732DSIEM/action/replication_record"}},"created_at":"2026-07-05T03:33:57.859767+00:00","updated_at":"2026-07-05T03:33:57.859767+00:00"}