{"id":"W2964212410","doi":"","title":"On calibration of modern neural networks","year":2017,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":1185,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Normalization (sociology); Computer science; Artificial neural network; Correctness; Calibration; Scaling; Deep neural networks; Artificial intelligence; Machine learning; Simple (philosophy); Deep learning; Pattern recognition (psychology); Data mining; Algorithm; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009801947,0.001546925,0.001199588,0.001528918,0.001077212,0.002423774,0.003124629,0.002455019,0.002555721],"category_scores_gemma":[0.06634749,0.0009977521,0.0008241342,0.001382558,0.003131785,0.005964175,0.003758067,0.006323694,0.000991137],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002778928,"about_ca_system_score_gemma":0.001476236,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004311454,"about_ca_topic_score_gemma":0.002731197,"domain_scores_codex":[0.9953661,0.001680136,0.0002116157,0.001100247,0.001346593,0.000295352],"domain_scores_gemma":[0.9818045,0.008971954,0.001533997,0.004284243,0.003115315,0.0002900253],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002700766,0.00007259173,0.005111555,0.0001335753,0.000153817,0.00009772449,0.0002767117,0.7474628,0.004691308,0.04942009,0.003786335,0.1885234],"study_design_scores_gemma":[0.00001227953,0.00004498034,0.0008077416,0.00005483561,0.00001593263,0.0000555208,0.0000295814,0.946363,0.003522083,0.04742347,0.001645011,0.00002551934],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0658917,0.002767731,0.9193286,0.002571022,0.0003057382,0.00006882412,0.0001854233,0.003188278,0.005692649],"genre_scores_gemma":[0.8029591,0.001706725,0.1886463,0.001376895,0.0003968954,0.0001998146,0.0005353052,0.0007356327,0.003443481],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009801947,"threshold_uncertainty_score":0.05183822,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03526583321291214,"score_gpt":0.3120534779422832,"score_spread":0.276787644729371,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}