{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:S6UKXG3ULNIVK4ANPJXMQ7LOD4","short_pith_number":"pith:S6UKXG3U","schema_version":"1.0","canonical_sha256":"97a8ab9b745b5155700d7a6ec87d6e1f0933a91bdbdc542806b4abd484bd7315","source":{"kind":"arxiv","id":"2108.05542","version":2},"attestation_state":"computed","paper":{"title":"AMMUS : A Survey of Transformer-based Pretrained Models in Natural Language Processing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ajit Rajasekharan, Katikapalli Subramanyam Kalyan, Sivanesan Sangeetha","submitted_at":"2021-08-12T05:32:18Z","abstract_excerpt":"Transformer-based pretrained language models (T-PTLMs) have achieved great success in almost every NLP task. The evolution of these models started with GPT and BERT. These models are built on the top of transformers, self-supervised learning and transfer learning. Transformed-based PTLMs learn universal language representations from large volumes of text data using self-supervised learning and transfer this knowledge to downstream tasks. These models provide good background knowledge to downstream tasks which avoids training of downstream models from scratch. In this comprehensive survey paper"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.05542","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-08-12T05:32:18Z","cross_cats_sorted":[],"title_canon_sha256":"6a28ed20e8440eb0595a6659bc39876b4edeeaa0abec4306ee41d87593ecbbc5","abstract_canon_sha256":"9753283f78833f5d77340278ce96c14efc84d40db50ba8a383b2be37a1d7d501"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:09:31.479846Z","signature_b64":"E92EldFIHO62OJpv9sXmUbNx4fuVvIs/WpSyjA/kjeeFOH42JvVWL8xOrBkYfxPprnH0ZN0bH+Ev2WPZDW9NBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"97a8ab9b745b5155700d7a6ec87d6e1f0933a91bdbdc542806b4abd484bd7315","last_reissued_at":"2026-07-05T03:09:31.479492Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:09:31.479492Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AMMUS : A Survey of Transformer-based Pretrained Models in Natural Language Processing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ajit Rajasekharan, Katikapalli Subramanyam Kalyan, Sivanesan Sangeetha","submitted_at":"2021-08-12T05:32:18Z","abstract_excerpt":"Transformer-based pretrained language models (T-PTLMs) have achieved great success in almost every NLP task. The evolution of these models started with GPT and BERT. These models are built on the top of transformers, self-supervised learning and transfer learning. Transformed-based PTLMs learn universal language representations from large volumes of text data using self-supervised learning and transfer this knowledge to downstream tasks. These models provide good background knowledge to downstream tasks which avoids training of downstream models from scratch. In this comprehensive survey paper"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.05542","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.05542/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.05542","created_at":"2026-07-05T03:09:31.479547+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.05542v2","created_at":"2026-07-05T03:09:31.479547+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.05542","created_at":"2026-07-05T03:09:31.479547+00:00"},{"alias_kind":"pith_short_12","alias_value":"S6UKXG3ULNIV","created_at":"2026-07-05T03:09:31.479547+00:00"},{"alias_kind":"pith_short_16","alias_value":"S6UKXG3ULNIVK4AN","created_at":"2026-07-05T03:09:31.479547+00:00"},{"alias_kind":"pith_short_8","alias_value":"S6UKXG3U","created_at":"2026-07-05T03:09:31.479547+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.03613","citing_title":"A Simple but Efficient Transformer-Based Physics-Informed Neural Network for Incompressible Navier--Stokes Equations","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2508.01486","citing_title":"Human-Centered Supervision for Sentiment Analysis in Telugu: A Systematic Inquiry Beyond Accuracy","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2211.14730","citing_title":"A Time Series is Worth 64 Words: Long-term Forecasting with Transformers","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S6UKXG3ULNIVK4ANPJXMQ7LOD4","json":"https://pith.science/pith/S6UKXG3ULNIVK4ANPJXMQ7LOD4.json","graph_json":"https://pith.science/api/pith-number/S6UKXG3ULNIVK4ANPJXMQ7LOD4/graph.json","events_json":"https://pith.science/api/pith-number/S6UKXG3ULNIVK4ANPJXMQ7LOD4/events.json","paper":"https://pith.science/paper/S6UKXG3U"},"agent_actions":{"view_html":"https://pith.science/pith/S6UKXG3ULNIVK4ANPJXMQ7LOD4","download_json":"https://pith.science/pith/S6UKXG3ULNIVK4ANPJXMQ7LOD4.json","view_paper":"https://pith.science/paper/S6UKXG3U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.05542&json=true","fetch_graph":"https://pith.science/api/pith-number/S6UKXG3ULNIVK4ANPJXMQ7LOD4/graph.json","fetch_events":"https://pith.science/api/pith-number/S6UKXG3ULNIVK4ANPJXMQ7LOD4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S6UKXG3ULNIVK4ANPJXMQ7LOD4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S6UKXG3ULNIVK4ANPJXMQ7LOD4/action/storage_attestation","attest_author":"https://pith.science/pith/S6UKXG3ULNIVK4ANPJXMQ7LOD4/action/author_attestation","sign_citation":"https://pith.science/pith/S6UKXG3ULNIVK4ANPJXMQ7LOD4/action/citation_signature","submit_replication":"https://pith.science/pith/S6UKXG3ULNIVK4ANPJXMQ7LOD4/action/replication_record"}},"created_at":"2026-07-05T03:09:31.479547+00:00","updated_at":"2026-07-05T03:09:31.479547+00:00"}