{"id":"W4384305065","doi":"10.1109/icse48619.2023.00008","title":"Artifact Evaluation","year":2023,"lang":"en","type":"article","venue":"","topic":"Industrial Vision Systems and Defect Detection","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal; University of Calgary; University of Victoria","funders":"","keywords":"Artifact (error); Computer science; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0003158898,0.00002996989,0.00003503824,0.00006784827,0.00002096253,0.00001426754,0.0000144273,0.00003601027,0.000389808],"category_scores_gemma":[0.00001984966,0.00002557527,0.00001798877,0.00024862,0.000001414059,0.00004259722,0.000002923469,0.00003102934,0.002957108],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002230982,"about_ca_system_score_gemma":0.000003628297,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000007622212,"about_ca_topic_score_gemma":0.00000436182,"domain_scores_codex":[0.9996825,0.00001128986,0.00007096898,0.00004074197,0.0001254657,0.00006899037],"domain_scores_gemma":[0.9998844,0.0000132781,0.00000390372,0.00006293981,0.00001885051,0.00001661477],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004814269,0.000004067228,0.0003082504,0.00001914869,0.00002999445,0.000002713524,0.0002005542,0.2301269,0.04184973,0.0004451103,0.1817792,0.5452294],"study_design_scores_gemma":[0.0002723767,0.00001877041,0.004534993,0.000007512177,0.000006167822,0.00000239512,0.0001350456,0.9246384,0.02460517,0.0003519289,0.04532243,0.0001048578],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8906011,0.00001357018,0.001658693,0.00001904511,0.00146766,0.0001675665,6.679554e-7,0.00154304,0.1045287],"genre_scores_gemma":[0.9991361,0.000002050565,0.000008571068,0.000004328274,0.0001218565,0.00001576441,0.000003320655,0.00000664591,0.0007012959],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6945114,"threshold_uncertainty_score":0.9978192,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05296498364968949,"score_gpt":0.2729635184845244,"score_spread":0.2199985348348349,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}