{"id":"W2121141437","doi":"10.1175/waf884.1","title":"Alternatives to the Chi-Square Test for Evaluating Rank Histograms from Ensemble Forecasts","year":2005,"lang":"en","type":"article","venue":"Weather and Forecasting","topic":"Forecasting Techniques and Applications","field":"Decision Sciences","cited_by":56,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Simon Fraser University","keywords":"Histogram; Rank (graph theory); Mathematics; Statistics; Statistical hypothesis testing; Econometrics; Artificial intelligence; Computer science; Combinatorics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03911324,0.001025502,0.002047003,0.006800281,0.001013057,0.00312538,0.002959501,0.0024808,0.007507589],"category_scores_gemma":[0.2843898,0.0004281731,0.001436893,0.007324373,0.002844868,0.005997256,0.002195643,0.003643682,0.001607464],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001217311,"about_ca_system_score_gemma":0.001580983,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002627261,"about_ca_topic_score_gemma":0.001700471,"domain_scores_codex":[0.9622177,0.02335211,0.002103983,0.002669301,0.008860754,0.0007961317],"domain_scores_gemma":[0.6197805,0.336004,0.009860238,0.01511157,0.01733707,0.001906693],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002651853,0.000457079,0.08239832,0.0007830641,0.001312087,0.0008888447,0.0009695493,0.1587707,0.003102404,0.2133422,0.01245365,0.5228704],"study_design_scores_gemma":[0.0001559075,0.001046408,0.02506317,0.0002127874,0.0001562159,0.0006326057,0.0008264522,0.8068671,0.004171528,0.1536224,0.006976681,0.0002687272],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09015834,0.001028978,0.9000084,0.0006964577,0.0004484233,0.000289899,0.0008669202,0.001063897,0.005438779],"genre_scores_gemma":[0.8099898,0.0004441763,0.1853996,0.0002398625,0.0004765738,0.0005594442,0.001184246,0.0002084321,0.001497848],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03911324,"threshold_uncertainty_score":0.206853,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2340837015960262,"score_gpt":0.4178356942781782,"score_spread":0.183751992682152,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}