{"id":"W4417334214","doi":"10.1177/18344909251406111","title":"Evaluating the Critical Thinking of Large Language Models: Insights and Limitations","year":2025,"lang":"en","type":"article","venue":"Journal of Pacific Rim Psychology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"National Natural Science Foundation of China","keywords":"Critical thinking; Argumentative; Cognition; Project commissioning; Key (lock); Publishing; Professional writing; Writing assessment","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03395009,0.0008010707,0.0007682878,0.001664123,0.0008063149,0.004526253,0.001985533,0.0005753203,0.001440869],"category_scores_gemma":[0.1494582,0.0003215846,0.0007011791,0.001301709,0.001698012,0.006718631,0.002755464,0.001265383,0.0004367077],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001496518,"about_ca_system_score_gemma":0.004337202,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004708813,"about_ca_topic_score_gemma":0.006072463,"domain_scores_codex":[0.9778774,0.01409423,0.001862872,0.001412146,0.00442213,0.000331259],"domain_scores_gemma":[0.8005794,0.1478479,0.01029446,0.01680204,0.02084414,0.003632123],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001443316,0.001706262,0.327412,0.002427313,0.0006134949,0.0003900455,0.02606476,0.01498436,0.008062562,0.0133695,0.00588068,0.5976456],"study_design_scores_gemma":[0.0007363051,0.005945203,0.3785469,0.003564206,0.0009539369,0.001393738,0.03912098,0.349975,0.03125361,0.1456373,0.04222669,0.0006461564],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9402241,0.002040449,0.04084482,0.003126928,0.000174229,0.0005575476,0.0003336156,0.0002420723,0.01245633],"genre_scores_gemma":[0.9780912,0.0003828187,0.02026416,0.0002479182,0.00003942239,0.0003504795,0.0001557548,0.00003162884,0.0004365238],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03395009,"threshold_uncertainty_score":0.1795474,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3430885522212014,"score_gpt":0.5717520442325003,"score_spread":0.2286634920112989,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}