{"id":"W4387592307","doi":"10.1017/9781316718636.021","title":"Evaluating Models","year":2023,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Figuring; Harm; Computer science; Psychology; Social psychology; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003781625,0.0003899773,0.0003807485,0.0003210066,0.0003441991,0.0001818633,0.002202329,0.0003381073,0.000002513112],"category_scores_gemma":[0.00003757335,0.000493056,0.0002380766,0.00003228296,0.0001565891,0.0004862311,0.001664952,0.0005222971,0.0002676639],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002652075,"about_ca_system_score_gemma":0.0002351054,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002226251,"about_ca_topic_score_gemma":0.000005131071,"domain_scores_codex":[0.9976441,0.00006373286,0.0002761947,0.0009197044,0.0006134326,0.0004828729],"domain_scores_gemma":[0.9976228,0.0002651526,0.0002564556,0.001297582,0.0003526111,0.0002053869],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0000106391,0.00000430169,5.471376e-8,0.00003409118,0.00005451928,0.0004022029,0.0001397189,0.0009423413,0.00005640125,0.9817339,0.008749619,0.007872282],"study_design_scores_gemma":[0.000322308,0.0002060315,0.000001153992,0.0004385881,0.0001543208,0.00002730874,0.00009350455,0.4252405,0.002704648,0.007494173,0.5618094,0.001508052],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.00002014767,0.0000431894,0.1177926,0.00003676908,0.000603991,0.0003381335,0.00003522866,0.0007288105,0.8804011],"genre_scores_gemma":[0.0003728632,0.0000561231,0.003178206,0.00006587321,0.0001183128,0.00000142408,0.00001147321,0.00005783678,0.9961379],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.9742396,"threshold_uncertainty_score":0.9997521,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1564211363363931,"score_gpt":0.2838641209800691,"score_spread":0.1274429846436761,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}