{"id":"W3041178285","doi":"10.3390/app10144704","title":"Interlaboratory Empirical Reproducibility Study Based on a GD&amp;T Benchmark","year":2020,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Advanced Measurement and Metrology Techniques","field":"Engineering","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"École de Technologie Supérieure","funders":"National Research Council Canada","keywords":"Benchmark (surveying); Computer science; Metrology; Engineering drawing; Engineering; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001253869,0.0001326319,0.000168758,0.00006815252,0.0001024269,0.00002260762,0.0002911882,0.00004150846,0.0000546443],"category_scores_gemma":[0.0002126745,0.0001100992,0.00002650341,0.0006237612,0.0001644881,0.0000648471,0.00003082975,0.000182759,0.00003135907],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003388043,"about_ca_system_score_gemma":0.0000345289,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":8.008321e-7,"about_ca_topic_score_gemma":0.000008010479,"domain_scores_codex":[0.9985684,0.00003394396,0.0001835236,0.0007026863,0.0002977837,0.000213669],"domain_scores_gemma":[0.9993439,0.0000631441,0.00002556743,0.0004532677,0.00002429019,0.0000898023],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0005303798,0.001737555,0.4868867,0.0001814278,0.0001143844,0.00002330151,0.009730832,0.06249104,0.3407491,0.001598493,0.05412886,0.04182789],"study_design_scores_gemma":[0.006085094,0.01017982,0.1852795,0.0001090044,0.0002397858,0.000002652939,0.008514707,0.06468048,0.5029664,0.009162699,0.2081841,0.004595719],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9536976,0.0000566769,0.01712974,0.0006117807,0.0001653891,0.0005691925,0.000002550202,0.000928544,0.02683851],"genre_scores_gemma":[0.9931316,0.000001198694,0.005561682,0.001149034,0.00007126489,0.0000705531,0.000001349604,0.000008843302,0.000004476246],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3016072,"threshold_uncertainty_score":0.4489717,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09920775398610127,"score_gpt":0.327189454600386,"score_spread":0.2279817006142847,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}