{"id":"W4400009090","doi":"10.20944/preprints202406.1635.v1","title":"Putting GPT-4o to the Sword: A Comprehensive Evaluation of Language, Vision, Speech, and Multimodal Proficiency","year":2024,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"SWORD; Computer science; Linguistics; Artificial intelligence; Natural language processing; Philosophy; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006841214,0.002190558,0.000998082,0.001595014,0.00056441,0.00299499,0.002386153,0.002307456,0.005730059],"category_scores_gemma":[0.03146222,0.0004453812,0.001252198,0.0007679079,0.001034427,0.005043615,0.005076616,0.002186355,0.002780525],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001275248,"about_ca_system_score_gemma":0.001836913,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007003657,"about_ca_topic_score_gemma":0.006625437,"domain_scores_codex":[0.9951579,0.002125327,0.0003395434,0.0009607246,0.001206472,0.0002100017],"domain_scores_gemma":[0.9859363,0.008082871,0.0005111538,0.002525659,0.001895248,0.001048821],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003958228,0.003898129,0.06695451,0.001767882,0.001332065,0.001006612,0.002313038,0.1661908,0.02429526,0.005264735,0.05333115,0.6696876],"study_design_scores_gemma":[0.0006026999,0.004821337,0.05157037,0.0004798926,0.000560482,0.0008649901,0.0018531,0.8624104,0.02974672,0.01317863,0.03354508,0.0003663536],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8066679,0.001557057,0.1264457,0.001735198,0.0006999537,0.001468179,0.01254919,0.0199665,0.02891042],"genre_scores_gemma":[0.8799908,0.0003733351,0.08126395,0.0007175267,0.00006704014,0.0008643253,0.02958084,0.0009791894,0.006163106],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007003657,"threshold_uncertainty_score":0.0361802,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.113106583943814,"score_gpt":0.3937495198749102,"score_spread":0.2806429359310962,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}