{"id":"W4400439180","doi":"10.1007/978-3-031-63800-8_11","title":"Investigating Calibrated Classification Scores Through the Lens of Interpretability","year":2024,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Forecasting Techniques and Applications","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Interpretability; Computer science; Information retrieval; Optometry; Artificial intelligence; Natural language processing; Medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.00339744,0.0001716131,0.0002586128,0.0005361922,0.0004341166,0.0006481431,0.003391796,0.0001170412,0.00002255088],"category_scores_gemma":[0.000643778,0.000115052,0.00007208774,0.001102219,0.003603817,0.00342989,0.002013614,0.0005060254,0.00005849533],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008179709,"about_ca_system_score_gemma":0.000271752,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006474704,"about_ca_topic_score_gemma":0.00003562115,"domain_scores_codex":[0.9972433,0.0000569132,0.001441771,0.0003171445,0.000795695,0.0001451653],"domain_scores_gemma":[0.9944883,0.001252482,0.0007239345,0.002701296,0.000790971,0.00004302262],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[7.594729e-7,0.000006159748,0.0001312295,0.00001517387,0.000002730171,1.27659e-8,0.002280999,0.00008437469,0.00001536213,0.950269,0.0006189207,0.04657523],"study_design_scores_gemma":[0.00004602965,0.00002772671,0.001778762,0.0002907269,0.000008884897,0.000005519367,0.0001542368,0.3853865,0.00003242571,0.4800195,0.132102,0.0001477113],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.001547742,0.0007518003,0.131326,0.01115343,0.0003535002,0.001260363,0.0001431027,0.000204236,0.8532599],"genre_scores_gemma":[0.934367,0.0008179044,0.0619939,0.0007937904,0.00003258211,0.00007808079,0.0000564341,0.00001450502,0.001845804],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9328192,"threshold_uncertainty_score":0.9991078,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2580863818663159,"score_gpt":0.4108652823804124,"score_spread":0.1527789005140965,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}