{"id":"W2162876064","doi":"10.1109/amuem.2005.1594618","title":"Key comparisons: applying the scientific method to validate uncertainty","year":2005,"lang":"en","type":"article","venue":"","topic":"Scientific Measurement and Uncertainty Evaluation","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Estimator; Consistency (knowledge bases); Key (lock); Statistics; Monte Carlo method; Computer science; Mean squared error; Degrees of freedom (physics and chemistry); Mathematics; Econometrics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.266414,0.001747165,0.003099909,0.01044438,0.002943354,0.008833864,0.004249761,0.004623256,0.007280305],"category_scores_gemma":[0.7237011,0.0007466868,0.003370531,0.01092442,0.01229621,0.01494496,0.006886722,0.004831478,0.0009202746],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003462936,"about_ca_system_score_gemma":0.006626906,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001857703,"about_ca_topic_score_gemma":0.001231715,"domain_scores_codex":[0.6636753,0.2609511,0.01355706,0.01717551,0.04280648,0.001834455],"domain_scores_gemma":[0.2484412,0.6306658,0.03008042,0.05543515,0.03416627,0.001211177],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001021571,0.0001345421,0.02350351,0.001499504,0.001620414,0.0004047095,0.003630568,0.02739808,0.001409749,0.7703089,0.005555723,0.1635128],"study_design_scores_gemma":[0.0002857564,0.0008774779,0.009908029,0.0008288543,0.0004399111,0.0005241474,0.001429237,0.1039623,0.005123018,0.8588684,0.01748607,0.0002668638],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02412643,0.0008827627,0.9614109,0.001833968,0.0005981659,0.0006191755,0.0004641525,0.0003412204,0.009723267],"genre_scores_gemma":[0.466951,0.0004211636,0.5279768,0.0006869863,0.0003529759,0.002153241,0.0003390086,0.0002293807,0.0008895114],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.733586,"threshold_uncertainty_score":0.9046421,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4080034840774005,"score_gpt":0.4853558283557723,"score_spread":0.07735234427837179,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}