{"id":"W2162876064","doi":"10.1109/amuem.2005.1594618","title":"Key comparisons: applying the scientific method to validate uncertainty","year":2005,"lang":"en","type":"article","venue":"","topic":"Scientific Measurement and Uncertainty Evaluation","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Estimator; Consistency (knowledge bases); Key (lock); Statistics; Monte Carlo method; Computer science; Mean squared error; Degrees of freedom (physics and chemistry); Mathematics; Econometrics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","sts","scholarly_communication","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.05309829,0.0001936359,0.0002821498,0.0004798265,0.001356466,0.002871427,0.001991504,0.00005388694,0.004924464],"category_scores_gemma":[0.003897565,0.0001018822,0.0001691181,0.00305525,0.0001806759,0.0004668987,0.0002713572,0.0001589585,0.007600706],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001254382,"about_ca_system_score_gemma":0.0001826502,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009048999,"about_ca_topic_score_gemma":0.0009365649,"domain_scores_codex":[0.9911602,0.001015917,0.0009479003,0.00105242,0.005295542,0.0005280118],"domain_scores_gemma":[0.9950763,0.001776556,0.0002380424,0.001490892,0.001173675,0.0002445617],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002943534,0.0000549358,0.001020543,9.127668e-7,0.00001345371,2.619033e-7,0.001772965,0.07162604,0.008011112,0.008658044,0.5358158,0.3729965],"study_design_scores_gemma":[0.000216066,0.00001602556,0.0007414478,0.000006577896,0.00001552304,0.000001770448,0.002226706,0.1776744,0.004817914,0.005787359,0.8083348,0.0001614393],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04469048,0.0001244237,0.8652822,0.02554623,0.002801427,0.002407001,0.00001286939,0.000176036,0.05895936],"genre_scores_gemma":[0.8237333,6.309232e-7,0.1094392,0.002950119,0.0003708914,0.0003244337,0.00001420329,0.00001476268,0.06315245],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7790428,"threshold_uncertainty_score":0.9999436,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4080034840774005,"score_gpt":0.4853558283557723,"score_spread":0.07735234427837179,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}