{"id":"W2736540189","doi":"10.7160/eriesj.2017.100202","title":"BAYESIAN DIAGNOSTICS FOR TEST DESIGN AND ANALYSIS","year":2017,"lang":"en","type":"article","venue":"Journal on Efficiency and Responsibility in Education and Science","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University; AXYS Technologies (Canada)","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Bayesian probability; Computer science; Bayesian statistics; Test (biology); Bridge (graph theory); Item response theory; Bayesian experimental design; Artificial intelligence; Machine learning; Posterior probability; Test theory; Bayesian inference; Econometrics; Natural language processing; Statistics; Mathematics; Psychometrics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1115107,0.002338054,0.003979551,0.007974132,0.001569149,0.006163416,0.004222347,0.00422933,0.006741961],"category_scores_gemma":[0.4901981,0.002015486,0.002024605,0.006229095,0.006274991,0.006045943,0.004909471,0.006988041,0.002443997],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003721781,"about_ca_system_score_gemma":0.008893658,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004311944,"about_ca_topic_score_gemma":0.00221613,"domain_scores_codex":[0.835936,0.1365736,0.004827987,0.005724091,0.01586227,0.001076026],"domain_scores_gemma":[0.5055214,0.4492579,0.01340553,0.01538492,0.01508541,0.001344853],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004363083,0.0002126103,0.007559876,0.001133159,0.000678119,0.0002529951,0.0008205921,0.05464344,0.0006433494,0.6635612,0.01012294,0.2599354],"study_design_scores_gemma":[0.0002269628,0.0002280235,0.001551344,0.0004668043,0.0001293201,0.0001585354,0.0001275202,0.2139179,0.0007830062,0.7709338,0.01138915,0.0000876651],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001078613,0.0004973834,0.9964343,0.0004517767,0.00008400377,0.0001926156,0.0001113767,0.0003760001,0.0007737775],"genre_scores_gemma":[0.07631542,0.0008716162,0.9170873,0.0005546118,0.0003499354,0.003053011,0.0005045629,0.0003141013,0.0009494631],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1115107,"threshold_uncertainty_score":0.5897319,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2680513155235794,"score_gpt":0.5076812887630598,"score_spread":0.2396299732394805,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}