{"id":"W4285223007","doi":"10.18653/v1/2022.acl-short.59","title":"Region-dependent temperature scaling for certainty calibration and application to class-imbalanced token classification","year":2022,"lang":"en","type":"article","venue":"","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph; National Research Council Canada; Vector Institute","funders":"","keywords":"Interpretability; Certainty; Calibration; Context (archaeology); Computer science; Metric (unit); Scaling; Class (philosophy); Range (aeronautics); Algorithm; Artificial intelligence; Mathematics; Statistics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002117957,0.00009671787,0.00009398869,0.0001021084,0.0006357264,0.0001401959,0.0003495312,0.00004891288,0.00000619269],"category_scores_gemma":[0.00001172939,0.00009787591,0.00003641073,0.0004030045,0.00001315265,0.0002260717,0.0001799963,0.0001099932,0.000001960251],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001148775,"about_ca_system_score_gemma":0.00003533637,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003049236,"about_ca_topic_score_gemma":0.00001035701,"domain_scores_codex":[0.9989716,0.0000383198,0.0002106013,0.000463317,0.0001701066,0.0001460635],"domain_scores_gemma":[0.9992712,0.00004642786,0.00009255733,0.0004284801,0.00008014616,0.00008115597],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003690339,0.0001081043,0.0003050647,0.0000197858,0.00001149578,4.166528e-7,0.0004923181,0.004848679,0.1006626,0.8047331,0.01040995,0.07837164],"study_design_scores_gemma":[0.0003119706,0.0001716829,0.001648944,0.000003403268,0.000007396835,0.00002815232,0.0003837722,0.9174145,0.01553216,0.01129489,0.05291184,0.0002913144],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008155714,0.00001344835,0.9804907,0.009352334,0.00005730779,0.001121849,0.00000870049,0.0003971685,0.0004027193],"genre_scores_gemma":[0.9513372,0.000006421098,0.04345202,0.001357142,0.00005555496,0.002961537,0.00003232521,0.00001011037,0.0007876936],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9431815,"threshold_uncertainty_score":0.4889558,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01434903487237385,"score_gpt":0.2557438075222285,"score_spread":0.2413947726498546,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}