{"id":"W3036485806","doi":"10.1037/xhp0000856","title":"Is zjudge a better prime for JUDGE than zudge is?: A new evaluation of current orthographic coding models.","year":2020,"lang":"en","type":"article","venue":"Journal of Experimental Psychology Human Perception & Performance","topic":"Reading and Literacy Development","field":"Psychology","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"Economic and Social Research Council; Natural Sciences and Engineering Research Council of Canada","keywords":"Orthographic projection; Coding (social sciences); Prime (order theory); Current (fluid); Computer science; Natural language processing; Artificial intelligence; Mathematics; Engineering; Statistics; Combinatorics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0008575273,0.0003152105,0.0005297125,0.0003948532,0.0001796309,0.00003522878,0.0004013764,0.0001815299,0.002731687],"category_scores_gemma":[0.000009732731,0.0002923283,0.0003831702,0.0002663952,0.0001372593,0.0005292861,0.0000434732,0.0004356002,0.0000713739],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001244747,"about_ca_system_score_gemma":0.00009666806,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000005893431,"about_ca_topic_score_gemma":5.199283e-7,"domain_scores_codex":[0.9970609,0.0001779523,0.001250436,0.0004834337,0.0006389965,0.000388265],"domain_scores_gemma":[0.9980951,0.00002576391,0.0009380315,0.0003247663,0.0003701271,0.0002462375],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.003108455,0.002216438,0.04429049,0.0001683136,0.0007647668,0.000007011339,0.2336616,0.0000613362,0.4000189,0.000707725,0.1692153,0.1457797],"study_design_scores_gemma":[0.04639833,0.01588,0.7913878,0.001425356,0.001149907,0.0008544424,0.01035969,0.0188634,0.04518077,0.002433653,0.06328128,0.00278538],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9900942,0.001891948,0.002853273,0.001369268,0.001249683,0.0005644235,0.0000148352,0.00002460989,0.001937764],"genre_scores_gemma":[0.9947008,0.0001339308,0.001280499,0.002928863,0.0006874741,0.00005855081,0.00002550147,0.00004272298,0.0001416698],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7470973,"threshold_uncertainty_score":0.9999529,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2192166758702141,"score_gpt":0.4501757246706802,"score_spread":0.2309590488004661,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}