{"id":"W2765850199","doi":"","title":"Methods for Classifying Errors on the Raven’s Standard Progressive Matrices Test - eScholarship","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Meeting of the Cognitive Science Society","topic":"Cognitive Abilities and Testing","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Raven's Progressive Matrices; Standard error; Test (biology); Analogy; Artificial intelligence; Cognition; Computer science; Psychology; Cognitive psychology; Statistics; Mathematics; Linguistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","sts"],"consensus_categories":["sts"],"category_scores_codex":[0.007045335,0.0002641121,0.0003285143,0.0000564016,0.001584714,0.0001893083,0.001699719,0.0001040946,0.00006724069],"category_scores_gemma":[0.0305217,0.0001329628,0.0005197846,0.001275219,0.003478302,0.0005161029,0.0007798053,0.0005494003,0.000005565171],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001027424,"about_ca_system_score_gemma":0.000159561,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005109379,"about_ca_topic_score_gemma":6.804618e-7,"domain_scores_codex":[0.9974332,0.0001228896,0.0005000828,0.0005689813,0.0007077236,0.0006670494],"domain_scores_gemma":[0.9849274,0.007829244,0.001415784,0.0002672036,0.005467067,0.00009326925],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"qualitative","study_design_scores_codex":[0.0005363934,0.001124574,0.3609832,0.001351628,0.0008113365,3.229509e-7,0.1919315,0.000009674491,0.1898931,0.03869831,0.00965818,0.2050018],"study_design_scores_gemma":[0.00111017,0.001024889,0.1277348,0.003183437,0.0003427855,0.000015174,0.517022,0.000483201,0.3024158,0.04573868,0.0003679052,0.000561194],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9792931,0.0001476802,0.00007259053,0.003159779,0.0004218726,0.002159229,0.0001229212,0.00004356919,0.01457926],"genre_scores_gemma":[0.9915032,0.000003585503,0.007024056,0.0006599685,0.0001169005,0.0003543562,3.357384e-7,0.00002835703,0.0003092335],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3250905,"threshold_uncertainty_score":0.9997151,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04865650131588967,"score_gpt":0.3758090381484613,"score_spread":0.3271525368325716,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}