{"id":"W4402193259","doi":"10.3390/app14177782","title":"Putting GPT-4o to the Sword: A Comprehensive Evaluation of Language, Vision, Speech, and Multimodal Proficiency","year":2024,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":107,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"SWORD; Linguistics; Psychology; Computer science; Philosophy; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007367642,0.002212527,0.001003789,0.001611157,0.0005828083,0.002944645,0.002294773,0.002086668,0.006007097],"category_scores_gemma":[0.03213914,0.0004046917,0.001341707,0.0007314382,0.001050936,0.004887534,0.004528154,0.002228,0.002476868],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001295102,"about_ca_system_score_gemma":0.002036376,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007397186,"about_ca_topic_score_gemma":0.007847016,"domain_scores_codex":[0.9956003,0.001961547,0.0003152983,0.000874198,0.001047901,0.0002006908],"domain_scores_gemma":[0.9868606,0.007877176,0.0004502349,0.002106886,0.001788604,0.0009165376],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003770849,0.003550157,0.06410345,0.00192558,0.00141455,0.001026053,0.001933359,0.1530349,0.02038504,0.005607618,0.04967962,0.6935688],"study_design_scores_gemma":[0.0005639779,0.00469297,0.04136529,0.0005779797,0.0005912188,0.0009373305,0.001858147,0.8818694,0.0237109,0.01436736,0.02910504,0.0003604364],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7724651,0.002310402,0.1606616,0.002247931,0.0009096269,0.001683784,0.01218365,0.01701144,0.03052656],"genre_scores_gemma":[0.8679655,0.0004725565,0.09638865,0.0009320576,0.00007899772,0.0008552429,0.02694328,0.0008649348,0.005498849],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007397186,"threshold_uncertainty_score":0.03896427,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1488573738870522,"score_gpt":0.4834961722266035,"score_spread":0.3346387983395512,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}