{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:M2EXCPNG52HB6EOCRVQF7WJYBL","short_pith_number":"pith:M2EXCPNG","schema_version":"1.0","canonical_sha256":"6689713da6ee8e1f11c28d605fd9380afc5f76ca76fd82ad1ccb52dd867b524e","source":{"kind":"arxiv","id":"2110.13473","version":2},"attestation_state":"computed","paper":{"title":"CTRN: Class-Temporal Relational Network for Action Detection","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Francois Bremond, Rui Dai, Srijan Das","submitted_at":"2021-10-26T08:15:47Z","abstract_excerpt":"Action detection is an essential and challenging task, especially for densely labelled datasets of untrimmed videos. There are many real-world challenges in those datasets, such as composite action, co-occurring action, and high temporal variation of instance duration. For handling these challenges, we propose to explore both the class and temporal relations of detected actions. In this work, we introduce an end-to-end network: Class-Temporal Relational Network (CTRN). It contains three key components: (1) The Representation Transform Module filters the class-specific features from the mixed r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.13473","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2021-10-26T08:15:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4f7b557ad0248380b0b3ac81ed75b2d72b59a6e5c0b4ea86751286077406fcfe","abstract_canon_sha256":"3c733d7b1083b1e68738832e819eb066c72a01b1af1bd0200c7cc4c40483bbda"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:38:45.601622Z","signature_b64":"9dR99BQ08JfJDRy32/lJX5eVEea9gPeAmIVDUb90hWzRLpGqsZumB7vRruO0aICz8qUOHZ78D8VNN2N25EasDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6689713da6ee8e1f11c28d605fd9380afc5f76ca76fd82ad1ccb52dd867b524e","last_reissued_at":"2026-07-05T04:38:45.601102Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:38:45.601102Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CTRN: Class-Temporal Relational Network for Action Detection","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Francois Bremond, Rui Dai, Srijan Das","submitted_at":"2021-10-26T08:15:47Z","abstract_excerpt":"Action detection is an essential and challenging task, especially for densely labelled datasets of untrimmed videos. There are many real-world challenges in those datasets, such as composite action, co-occurring action, and high temporal variation of instance duration. For handling these challenges, we propose to explore both the class and temporal relations of detected actions. In this work, we introduce an end-to-end network: Class-Temporal Relational Network (CTRN). It contains three key components: (1) The Representation Transform Module filters the class-specific features from the mixed r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.13473","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.13473/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.13473","created_at":"2026-07-05T04:38:45.601158+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.13473v2","created_at":"2026-07-05T04:38:45.601158+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.13473","created_at":"2026-07-05T04:38:45.601158+00:00"},{"alias_kind":"pith_short_12","alias_value":"M2EXCPNG52HB","created_at":"2026-07-05T04:38:45.601158+00:00"},{"alias_kind":"pith_short_16","alias_value":"M2EXCPNG52HB6EOC","created_at":"2026-07-05T04:38:45.601158+00:00"},{"alias_kind":"pith_short_8","alias_value":"M2EXCPNG","created_at":"2026-07-05T04:38:45.601158+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.14505","citing_title":"LLaVA-MR: Large Language-and-Vision Assistant for Video Moment Retrieval","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M2EXCPNG52HB6EOCRVQF7WJYBL","json":"https://pith.science/pith/M2EXCPNG52HB6EOCRVQF7WJYBL.json","graph_json":"https://pith.science/api/pith-number/M2EXCPNG52HB6EOCRVQF7WJYBL/graph.json","events_json":"https://pith.science/api/pith-number/M2EXCPNG52HB6EOCRVQF7WJYBL/events.json","paper":"https://pith.science/paper/M2EXCPNG"},"agent_actions":{"view_html":"https://pith.science/pith/M2EXCPNG52HB6EOCRVQF7WJYBL","download_json":"https://pith.science/pith/M2EXCPNG52HB6EOCRVQF7WJYBL.json","view_paper":"https://pith.science/paper/M2EXCPNG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.13473&json=true","fetch_graph":"https://pith.science/api/pith-number/M2EXCPNG52HB6EOCRVQF7WJYBL/graph.json","fetch_events":"https://pith.science/api/pith-number/M2EXCPNG52HB6EOCRVQF7WJYBL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M2EXCPNG52HB6EOCRVQF7WJYBL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M2EXCPNG52HB6EOCRVQF7WJYBL/action/storage_attestation","attest_author":"https://pith.science/pith/M2EXCPNG52HB6EOCRVQF7WJYBL/action/author_attestation","sign_citation":"https://pith.science/pith/M2EXCPNG52HB6EOCRVQF7WJYBL/action/citation_signature","submit_replication":"https://pith.science/pith/M2EXCPNG52HB6EOCRVQF7WJYBL/action/replication_record"}},"created_at":"2026-07-05T04:38:45.601158+00:00","updated_at":"2026-07-05T04:38:45.601158+00:00"}