{"id":"W4412697139","doi":"10.1007/978-3-031-87869-5_2","title":"Evaluation Models and their Implementation","year":2025,"lang":"en","type":"book-chapter","venue":"","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Toronto East General Hospital; Trillium Health Centre; University of Toronto","funders":"","keywords":"Computer science; Management science; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01345003,0.001194584,0.001143365,0.002648775,0.001228824,0.008852541,0.002017341,0.002435078,0.01059164],"category_scores_gemma":[0.02720873,0.00109718,0.0009245392,0.002398545,0.005253601,0.008565281,0.002605712,0.003369527,0.003107984],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00427321,"about_ca_system_score_gemma":0.004465386,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003848579,"about_ca_topic_score_gemma":0.003362374,"domain_scores_codex":[0.9881842,0.007978751,0.0004171089,0.000545534,0.002647403,0.0002269056],"domain_scores_gemma":[0.9879755,0.009279847,0.0002609184,0.001108362,0.001243334,0.0001320184],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000003115013,0.00001537301,0.00004316558,0.0000575851,0.00000644332,0.000008788759,0.0001050318,0.00377442,0.00002219045,0.9532066,0.005505977,0.03725137],"study_design_scores_gemma":[0.000003231621,0.000008160751,0.00005281084,0.0001534262,0.000006352339,0.00002027388,0.0001015604,0.01128971,0.0001110916,0.9419446,0.04629965,0.000009045965],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001913484,0.008505343,0.7096765,0.01064324,0.0007486375,0.0002196232,0.0001580998,0.0006678957,0.2674672],"genre_scores_gemma":[0.19527,0.01734627,0.6511536,0.00241171,0.0008451474,0.001405413,0.0004138985,0.0006385965,0.1305154],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01345003,"threshold_uncertainty_score":0.07113147,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4519174322867391,"score_gpt":0.5626042380772623,"score_spread":0.1106868057905233,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}