{"id":"W3041178285","doi":"10.3390/app10144704","title":"Interlaboratory Empirical Reproducibility Study Based on a GD&amp;T Benchmark","year":2020,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Advanced Measurement and Metrology Techniques","field":"Engineering","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"École de Technologie Supérieure","funders":"National Research Council Canada","keywords":"Benchmark (surveying); Computer science; Metrology; Engineering drawing; Engineering; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04465877,0.001084172,0.0008708294,0.002527306,0.00127374,0.002002684,0.001595422,0.001639874,0.001138822],"category_scores_gemma":[0.06944063,0.0003240368,0.00164762,0.002943539,0.001936924,0.001165814,0.002189905,0.000840502,0.0007095145],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00112488,"about_ca_system_score_gemma":0.001415333,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003076136,"about_ca_topic_score_gemma":0.00282266,"domain_scores_codex":[0.9539374,0.01628812,0.003967142,0.009949611,0.01520911,0.0006485363],"domain_scores_gemma":[0.8856638,0.05198811,0.009199512,0.01914447,0.03285768,0.001146374],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00375941,0.003463285,0.6951048,0.002024346,0.002389903,0.0009438157,0.0079623,0.02067108,0.09165017,0.003793795,0.005523004,0.1627141],"study_design_scores_gemma":[0.0003459248,0.01616181,0.7766885,0.0004537741,0.002319125,0.002328234,0.005176508,0.05657148,0.1045062,0.003408054,0.03163373,0.0004066555],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8816621,0.003252141,0.1023703,0.0002135812,0.0003220699,0.001071501,0.001556054,0.0005105163,0.009041728],"genre_scores_gemma":[0.9714558,0.0002973255,0.02410431,0.0001694418,0.00008861862,0.0006089057,0.002022108,0.0001136432,0.001139863],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9553412,"threshold_uncertainty_score":0.236181,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09920775398610127,"score_gpt":0.327189454600386,"score_spread":0.2279817006142847,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}