{"id":"W2765850199","doi":"","title":"Methods for Classifying Errors on the Raven’s Standard Progressive Matrices Test - eScholarship","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Meeting of the Cognitive Science Society","topic":"Cognitive Abilities and Testing","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Raven's Progressive Matrices; Standard error; Test (biology); Analogy; Artificial intelligence; Cognition; Computer science; Psychology; Cognitive psychology; Statistics; Mathematics; Linguistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0219342,0.002610061,0.001440037,0.01047858,0.001047797,0.00287875,0.003415937,0.001762278,0.0106332],"category_scores_gemma":[0.1010909,0.0008121754,0.002006974,0.004174076,0.001105997,0.002967704,0.002570757,0.002132453,0.004606965],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007169406,"about_ca_system_score_gemma":0.001844679,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003406879,"about_ca_topic_score_gemma":0.005703519,"domain_scores_codex":[0.97496,0.005924124,0.005618605,0.002957932,0.009920213,0.0006190597],"domain_scores_gemma":[0.939701,0.01882911,0.0126783,0.007434815,0.02047656,0.0008801953],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001539472,0.0006823137,0.200996,0.001132308,0.0007451618,0.000342466,0.002554703,0.003344859,0.007887358,0.01094974,0.02736032,0.7424653],"study_design_scores_gemma":[0.001002961,0.002990208,0.6992579,0.002027197,0.0008940815,0.004908022,0.003359119,0.06490312,0.04841835,0.05879173,0.1122776,0.001169722],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.106395,0.001802326,0.8479033,0.0008142001,0.0006181886,0.009946221,0.0088825,0.005267892,0.0183703],"genre_scores_gemma":[0.1624333,0.0009714643,0.7941148,0.000347875,0.0002073771,0.01883375,0.006556913,0.001521787,0.01501271],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0219342,"threshold_uncertainty_score":0.1160005,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04865650131588967,"score_gpt":0.3758090381484613,"score_spread":0.3271525368325716,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}