{"id":"W4400439180","doi":"10.1007/978-3-031-63800-8_11","title":"Investigating Calibrated Classification Scores Through the Lens of Interpretability","year":2024,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Forecasting Techniques and Applications","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Interpretability; Computer science; Information retrieval; Optometry; Artificial intelligence; Natural language processing; Medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0240414,0.001142769,0.001076183,0.004109057,0.0008375431,0.01245126,0.002501214,0.002668933,0.007010259],"category_scores_gemma":[0.2012362,0.0006467667,0.0007748876,0.004081106,0.007498992,0.01373708,0.003338812,0.005741434,0.0006736806],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003006252,"about_ca_system_score_gemma":0.00124486,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001865851,"about_ca_topic_score_gemma":0.001198096,"domain_scores_codex":[0.9799493,0.01352896,0.000671463,0.002049707,0.003303486,0.0004970692],"domain_scores_gemma":[0.8817561,0.09455621,0.007231788,0.01047423,0.005466339,0.0005153349],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008147225,0.00005376788,0.007570426,0.0001552537,0.000117252,0.0001308615,0.002349552,0.0170991,0.001523494,0.8806078,0.002084246,0.08822674],"study_design_scores_gemma":[0.00001338894,0.00003252694,0.003639919,0.00008096106,0.00002877723,0.00006005211,0.0006184193,0.05614907,0.001056271,0.936555,0.001734837,0.00003072062],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1584121,0.001833269,0.7554039,0.01232141,0.0002764834,0.0000867321,0.0004446466,0.0007251831,0.07049631],"genre_scores_gemma":[0.9015962,0.00050204,0.09396159,0.0004971442,0.0002648705,0.00008809542,0.0003001849,0.0003485258,0.002441423],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0240414,"threshold_uncertainty_score":0.1271446,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2580863818663159,"score_gpt":0.4108652823804124,"score_spread":0.1527789005140965,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}