{"id":"W4406883428","doi":"10.1007/978-3-031-81068-8_11","title":"Learning from Mistakes","year":2025,"lang":"en","type":"book-chapter","venue":"","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University","funders":"","keywords":"Psychology; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002616805,0.0009649231,0.0005773792,0.001097306,0.001775189,0.006465627,0.001543571,0.002213727,0.03729714],"category_scores_gemma":[0.01215501,0.0003857529,0.0003766557,0.0007358866,0.005459311,0.01232564,0.00409857,0.00489181,0.02074877],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002090388,"about_ca_system_score_gemma":0.002770952,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002124217,"about_ca_topic_score_gemma":0.004778273,"domain_scores_codex":[0.9976832,0.0007409846,0.00006358943,0.0002491648,0.001112288,0.0001507881],"domain_scores_gemma":[0.995903,0.001968737,0.0001493497,0.0007046313,0.0009986546,0.0002756203],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001603025,0.00004322561,0.0003147726,0.000125184,0.000009002192,0.0001081385,0.002224879,0.0007852737,0.0002369163,0.4561344,0.2528742,0.2871279],"study_design_scores_gemma":[0.000003328159,0.00001583928,0.0002372748,0.0003858921,0.000004911893,0.0001639877,0.001129211,0.0008627954,0.0003731608,0.463241,0.53357,0.00001252678],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.001854693,0.00445471,0.04198728,0.015486,0.001387101,0.00004405712,0.00008404438,0.000374508,0.9343276],"genre_scores_gemma":[0.04859724,0.005063157,0.02188373,0.006685419,0.0007701544,0.00009692609,0.00024939,0.0004817605,0.9161721],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.03729714,"threshold_uncertainty_score":0.1247714,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2827184533552689,"score_gpt":0.4703562651379598,"score_spread":0.1876378117826909,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}