{"id":"W4401313066","doi":"10.18260/1-2--48322","title":"WIP: Traditional Engineering Assessments Challenged by ChatGPT: An Evaluation of its Performance on a Fundamental Competencies Exam","year":2024,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Context (archaeology); Coherence (philosophical gambling strategy); Relevance (law); Computer science; Inclusion (mineral); Engineering education; Function (biology); Mathematics education; Engineering management; Psychology; Engineering; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004698243,0.000920996,0.0008856264,0.001720282,0.0004142051,0.001253899,0.001009569,0.001333945,0.004220028],"category_scores_gemma":[0.02419359,0.0002457178,0.0005898381,0.0006760164,0.0004324996,0.001239732,0.001970672,0.0009235528,0.00269713],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006556113,"about_ca_system_score_gemma":0.0006544854,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002241914,"about_ca_topic_score_gemma":0.0017864,"domain_scores_codex":[0.9956661,0.001844126,0.0004032349,0.0006552946,0.001157492,0.0002737187],"domain_scores_gemma":[0.9831887,0.008442935,0.001348412,0.001302328,0.002749554,0.002968133],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.01545073,0.01707072,0.3240946,0.001927749,0.0005894155,0.001660687,0.009086612,0.03106831,0.03632038,0.001006896,0.01708911,0.5446348],"study_design_scores_gemma":[0.0006806409,0.02783464,0.765624,0.0002703027,0.000434154,0.0008993316,0.007105823,0.1444578,0.03292108,0.001049821,0.01838878,0.0003336151],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9924483,0.0001086657,0.003030302,0.00009025328,0.00006005098,0.0002169505,0.0005292928,0.000539955,0.002976178],"genre_scores_gemma":[0.9904838,0.00007255647,0.004893762,0.00005459109,0.00003258148,0.0002394445,0.001398197,0.00005965885,0.002765481],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004698243,"threshold_uncertainty_score":0.02484697,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4235209335340918,"score_gpt":0.463156249067314,"score_spread":0.03963531553322219,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}