{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IXM7KTJKJ33UW5GRHOTC3QPXTL","short_pith_number":"pith:IXM7KTJK","schema_version":"1.0","canonical_sha256":"45d9f54d2a4ef74b74d13ba62dc1f79ad9f22b73711c91bd9e03dc2f4eb39984","source":{"kind":"arxiv","id":"2405.02771","version":2},"attestation_state":"computed","paper":{"title":"MMEarth: Exploring Multi-Modal Pretext Tasks For Geospatial Representation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Ankit Kariryaa, Christian Igel, Nico Lang, Serge Belongie, Stefan Oehmcke, Vishal Nedungadi","submitted_at":"2024-05-04T23:16:48Z","abstract_excerpt":"The volume of unlabelled Earth observation (EO) data is huge, but many important applications lack labelled training data. However, EO data offers the unique opportunity to pair data from different modalities and sensors automatically based on geographic location and time, at virtually no human labor cost. We seize this opportunity to create MMEarth, a diverse multi-modal pretraining dataset at global scale. Using this new corpus of 1.2 million locations, we propose a Multi-Pretext Masked Autoencoder (MP-MAE) approach to learn general-purpose representations for optical satellite images. Our a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.02771","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-04T23:16:48Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"2800826a2273e357d47753c764cb54ea989ff02f4d89de8e5c4c867c5ba7b012","abstract_canon_sha256":"ba3d4f1d7d8e44ed74899a76f198ae70e61b3054f6ae4a1a377d1e9fd2052269"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:34.117658Z","signature_b64":"yTc0Ln+MPw0IXa7SCjYBBEz+z/58HIRf8DWgsCbreEWI/2bjwjqFfNajISCoLNlSI9xDHJxMyKeJGrr0i3OcAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45d9f54d2a4ef74b74d13ba62dc1f79ad9f22b73711c91bd9e03dc2f4eb39984","last_reissued_at":"2026-07-05T08:49:34.117048Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:34.117048Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMEarth: Exploring Multi-Modal Pretext Tasks For Geospatial Representation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Ankit Kariryaa, Christian Igel, Nico Lang, Serge Belongie, Stefan Oehmcke, Vishal Nedungadi","submitted_at":"2024-05-04T23:16:48Z","abstract_excerpt":"The volume of unlabelled Earth observation (EO) data is huge, but many important applications lack labelled training data. However, EO data offers the unique opportunity to pair data from different modalities and sensors automatically based on geographic location and time, at virtually no human labor cost. We seize this opportunity to create MMEarth, a diverse multi-modal pretraining dataset at global scale. Using this new corpus of 1.2 million locations, we propose a Multi-Pretext Masked Autoencoder (MP-MAE) approach to learn general-purpose representations for optical satellite images. Our a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.02771","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.02771/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.02771","created_at":"2026-07-05T08:49:34.117133+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.02771v2","created_at":"2026-07-05T08:49:34.117133+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.02771","created_at":"2026-07-05T08:49:34.117133+00:00"},{"alias_kind":"pith_short_12","alias_value":"IXM7KTJKJ33U","created_at":"2026-07-05T08:49:34.117133+00:00"},{"alias_kind":"pith_short_16","alias_value":"IXM7KTJKJ33UW5GR","created_at":"2026-07-05T08:49:34.117133+00:00"},{"alias_kind":"pith_short_8","alias_value":"IXM7KTJK","created_at":"2026-07-05T08:49:34.117133+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07758","citing_title":"Scalable and Trustworthy Earth Observation Foundation Models","ref_index":48,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20523","citing_title":"SARLO-80: Worldwide Slant SAR Language Optic Dataset 80cm","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2503.19719","citing_title":"On What Depends the Robustness of Multi-source Models to Missing Data in Earth Observation?","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IXM7KTJKJ33UW5GRHOTC3QPXTL","json":"https://pith.science/pith/IXM7KTJKJ33UW5GRHOTC3QPXTL.json","graph_json":"https://pith.science/api/pith-number/IXM7KTJKJ33UW5GRHOTC3QPXTL/graph.json","events_json":"https://pith.science/api/pith-number/IXM7KTJKJ33UW5GRHOTC3QPXTL/events.json","paper":"https://pith.science/paper/IXM7KTJK"},"agent_actions":{"view_html":"https://pith.science/pith/IXM7KTJKJ33UW5GRHOTC3QPXTL","download_json":"https://pith.science/pith/IXM7KTJKJ33UW5GRHOTC3QPXTL.json","view_paper":"https://pith.science/paper/IXM7KTJK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.02771&json=true","fetch_graph":"https://pith.science/api/pith-number/IXM7KTJKJ33UW5GRHOTC3QPXTL/graph.json","fetch_events":"https://pith.science/api/pith-number/IXM7KTJKJ33UW5GRHOTC3QPXTL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IXM7KTJKJ33UW5GRHOTC3QPXTL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IXM7KTJKJ33UW5GRHOTC3QPXTL/action/storage_attestation","attest_author":"https://pith.science/pith/IXM7KTJKJ33UW5GRHOTC3QPXTL/action/author_attestation","sign_citation":"https://pith.science/pith/IXM7KTJKJ33UW5GRHOTC3QPXTL/action/citation_signature","submit_replication":"https://pith.science/pith/IXM7KTJKJ33UW5GRHOTC3QPXTL/action/replication_record"}},"created_at":"2026-07-05T08:49:34.117133+00:00","updated_at":"2026-07-05T08:49:34.117133+00:00"}