{"id":"W4391998433","doi":"10.53841/bpsadm.2024.16.1.48","title":"Large-scale testing in the face of AI","year":2024,"lang":"en","type":"article","venue":"Assessment and Development Matters","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"First Nations University of Canada; Brock University","funders":"","keywords":"Expansive; Test (biology); Computer science; Construct (python library); Scale (ratio); Face validity; Set (abstract data type); Test design; Face (sociological concept); Construct validity; Data science; Psychology; Political science; Sociology; Psychometrics; Social science; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.203934,0.0006064388,0.0007616223,0.001499257,0.002295987,0.005980673,0.004641755,0.002349364,0.009974803],"category_scores_gemma":[0.3846985,0.0003984929,0.0005401369,0.0009653433,0.007835096,0.009601115,0.00714049,0.004986406,0.00154151],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003886333,"about_ca_system_score_gemma":0.009955844,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001595855,"about_ca_topic_score_gemma":0.00301162,"domain_scores_codex":[0.8320197,0.13832,0.003757276,0.004577769,0.01963812,0.0016871],"domain_scores_gemma":[0.3490077,0.5330047,0.01334712,0.0574923,0.03912006,0.008028042],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0008369401,0.001894818,0.03799508,0.001399291,0.0001485046,0.0012006,0.01658512,0.008458165,0.005987276,0.1047449,0.03468578,0.7860635],"study_design_scores_gemma":[0.0006234025,0.007439803,0.08318117,0.005077696,0.0001906332,0.004353817,0.04319122,0.05746017,0.01731337,0.5656489,0.215137,0.0003828592],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2395727,0.003914845,0.5133101,0.1504171,0.002534039,0.005832901,0.0004431011,0.003392807,0.08058237],"genre_scores_gemma":[0.8051476,0.0005782876,0.1791298,0.007441495,0.0006659002,0.002395849,0.0002130089,0.0002778608,0.004150094],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.203934,"threshold_uncertainty_score":0.9816912,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1316125422972995,"score_gpt":0.4430557148226341,"score_spread":0.3114431725253346,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}