{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:63QVOCCVIN4M3M7TQBDX736DCP","short_pith_number":"pith:63QVOCCV","schema_version":"1.0","canonical_sha256":"f6e15708554378cdb3f380477fefc313dcc3fc3da98c29bdcebc99c303adaba0","source":{"kind":"arxiv","id":"1801.04062","version":5},"attestation_state":"computed","paper":{"title":"MINE: Mutual Information Neural Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Aaron Courville, Aristide Baratin, Mohamed Ishmael Belghazi, R Devon Hjelm, Sai Rajeswar, Sherjil Ozair, Yoshua Bengio","submitted_at":"2018-01-12T05:42:58Z","abstract_excerpt":"We argue that the estimation of mutual information between high dimensional continuous random variables can be achieved by gradient descent over neural networks. We present a Mutual Information Neural Estimator (MINE) that is linearly scalable in dimensionality as well as in sample size, trainable through back-prop, and strongly consistent. We present a handful of applications on which MINE can be used to minimize or maximize mutual information. We apply MINE to improve adversarially trained generative models. We also use MINE to implement Information Bottleneck, applying it to supervised clas"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1801.04062","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-01-12T05:42:58Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"d972655c180c4bdf338ca892f0fc4c3f6d34ed9665bb0fd0e9d5386c083ef0bc","abstract_canon_sha256":"b55909e843571715ee98d2d11e933a8cee98b1d32cdb91c893ad83d063435ad6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:05:40.353296Z","signature_b64":"AaCgh847FJoEpSsteCCOdQQNJiBa35Q7cMY17CycncqMwlrqYsYvJxarsJv0tCfZ4zjytjXupI+3WAgvXYxICw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f6e15708554378cdb3f380477fefc313dcc3fc3da98c29bdcebc99c303adaba0","last_reissued_at":"2026-07-05T03:05:40.352878Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:05:40.352878Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MINE: Mutual Information Neural Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Aaron Courville, Aristide Baratin, Mohamed Ishmael Belghazi, R Devon Hjelm, Sai Rajeswar, Sherjil Ozair, Yoshua Bengio","submitted_at":"2018-01-12T05:42:58Z","abstract_excerpt":"We argue that the estimation of mutual information between high dimensional continuous random variables can be achieved by gradient descent over neural networks. We present a Mutual Information Neural Estimator (MINE) that is linearly scalable in dimensionality as well as in sample size, trainable through back-prop, and strongly consistent. We present a handful of applications on which MINE can be used to minimize or maximize mutual information. We apply MINE to improve adversarially trained generative models. We also use MINE to implement Information Bottleneck, applying it to supervised clas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1801.04062","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1801.04062/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1801.04062","created_at":"2026-07-05T03:05:40.352945+00:00"},{"alias_kind":"arxiv_version","alias_value":"1801.04062v5","created_at":"2026-07-05T03:05:40.352945+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1801.04062","created_at":"2026-07-05T03:05:40.352945+00:00"},{"alias_kind":"pith_short_12","alias_value":"63QVOCCVIN4M","created_at":"2026-07-05T03:05:40.352945+00:00"},{"alias_kind":"pith_short_16","alias_value":"63QVOCCVIN4M3M7T","created_at":"2026-07-05T03:05:40.352945+00:00"},{"alias_kind":"pith_short_8","alias_value":"63QVOCCV","created_at":"2026-07-05T03:05:40.352945+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.05175","citing_title":"Platonic Projection Structures: Operator-Induced Observability in Representation Learning","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2604.17709","citing_title":"DeInfer: Efficient Parallel Inferencing for Decomposed Large Language Models","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26091","citing_title":"On-Policy Self-Distillation with Sampled Demonstrations Reduces Output Diversity","ref_index":188,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23611","citing_title":"Data Selection Through Iterative Self-Filtering for Vision-Language Settings","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05561","citing_title":"InfoShield: Privacy-Preserving Speech Representations for Mental Health Screening via Information-Theoretic Optimization","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"1907.00348","citing_title":"Learning to Find Correlated Features by Maximizing Information Flow in Convolutional Neural Networks","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2408.13122","citing_title":"Semantic Variational Bayes Based on Semantic Information G Theory for Solving Latent Variables","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01202","citing_title":"Forget BIT, It is All about TOKEN: Towards Semantic Information Theory for LLMs","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11170","citing_title":"Unlearning with Asymmetric Sources: Improved Unlearning-Utility Trade-off with Public Data","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03413","citing_title":"Learning to Theorize the World from Observation","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17710","citing_title":"Dynamic Visual-semantic Alignment for Zero-shot Learning with Ambiguous Labels","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03039","citing_title":"Mixed-Precision Information Bottlenecks for On-Device Trait-State Disentanglement in Bipolar Agitation Detection","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/63QVOCCVIN4M3M7TQBDX736DCP","json":"https://pith.science/pith/63QVOCCVIN4M3M7TQBDX736DCP.json","graph_json":"https://pith.science/api/pith-number/63QVOCCVIN4M3M7TQBDX736DCP/graph.json","events_json":"https://pith.science/api/pith-number/63QVOCCVIN4M3M7TQBDX736DCP/events.json","paper":"https://pith.science/paper/63QVOCCV"},"agent_actions":{"view_html":"https://pith.science/pith/63QVOCCVIN4M3M7TQBDX736DCP","download_json":"https://pith.science/pith/63QVOCCVIN4M3M7TQBDX736DCP.json","view_paper":"https://pith.science/paper/63QVOCCV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1801.04062&json=true","fetch_graph":"https://pith.science/api/pith-number/63QVOCCVIN4M3M7TQBDX736DCP/graph.json","fetch_events":"https://pith.science/api/pith-number/63QVOCCVIN4M3M7TQBDX736DCP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/63QVOCCVIN4M3M7TQBDX736DCP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/63QVOCCVIN4M3M7TQBDX736DCP/action/storage_attestation","attest_author":"https://pith.science/pith/63QVOCCVIN4M3M7TQBDX736DCP/action/author_attestation","sign_citation":"https://pith.science/pith/63QVOCCVIN4M3M7TQBDX736DCP/action/citation_signature","submit_replication":"https://pith.science/pith/63QVOCCVIN4M3M7TQBDX736DCP/action/replication_record"}},"created_at":"2026-07-05T03:05:40.352945+00:00","updated_at":"2026-07-05T03:05:40.352945+00:00"}