{"id":"W4411687687","doi":"10.1109/tii.2025.3577686","title":"TransCLIP: Transferring Vision–Language Models for Efficient Video Action Recognition","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Industrial Informatics","topic":"Human Pose and Action Recognition","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"China Postdoctoral Science Foundation","keywords":"Computer science; Action recognition; Computer vision; Action (physics); Artificial intelligence; Speech recognition","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.000425381,0.0002511205,0.0002594268,0.0006769554,0.0005570528,0.0003044244,0.0003422686,0.000319089,0.00004728932],"category_scores_gemma":[0.00001106886,0.0002557469,0.0002747977,0.0007003204,0.00004087887,0.001379927,0.000001877235,0.0005031927,0.00005909716],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001719567,"about_ca_system_score_gemma":0.0001534172,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002177039,"about_ca_topic_score_gemma":0.00001918957,"domain_scores_codex":[0.9982209,0.00006313792,0.000806483,0.0002456772,0.0003155379,0.0003482151],"domain_scores_gemma":[0.9989649,0.0002726309,0.0001347759,0.000357284,0.0001626922,0.0001077136],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002046892,0.0003250267,8.5764e-8,0.0001028302,0.000091965,9.193201e-7,0.003098127,0.1455113,0.0007059856,0.0009539734,0.0007918085,0.8482133],"study_design_scores_gemma":[0.003598108,0.0003545074,0.000001194814,0.0003430631,0.0001165802,0.000009818667,0.001150231,0.8457209,0.1440477,0.002587884,0.001681273,0.0003887497],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03613051,0.000007213945,0.9574139,0.0003672426,0.002441715,0.0009701945,0.0001014658,0.0003718994,0.002195884],"genre_scores_gemma":[0.9911523,0.00003389028,0.007395337,0.0005967876,0.0001298296,0.0002795028,0.00003196644,0.00001680221,0.0003636068],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9550218,"threshold_uncertainty_score":0.9999894,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0812729455304916,"score_gpt":0.3095092651723503,"score_spread":0.2282363196418587,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}