{"id":"W4396629499","doi":"10.1109/ms.2024.3395616","title":"How Trustworthy Is Your Continuous Integration (CI) Accelerator?: A Comparison of the Trustworthiness of CI Acceleration Products","year":2024,"lang":"en","type":"article","venue":"IEEE Software","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal; University of Waterloo","funders":"","keywords":"Trustworthiness; Acceleration; Computer science; Software engineering; Computer security; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04672338,0.0006781268,0.000597952,0.003388413,0.0009038334,0.003352372,0.00131299,0.001001736,0.0009610801],"category_scores_gemma":[0.2680737,0.0007235914,0.000592615,0.002723218,0.002323126,0.004168856,0.003126909,0.001538575,0.0006162211],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001336988,"about_ca_system_score_gemma":0.001766467,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00365079,"about_ca_topic_score_gemma":0.002779537,"domain_scores_codex":[0.9653329,0.0116052,0.002818365,0.003415954,0.01564205,0.001185542],"domain_scores_gemma":[0.6345499,0.208749,0.05093738,0.04535989,0.05401219,0.006391517],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003397834,0.0004320463,0.6787901,0.0006464816,0.0006045225,0.0008069603,0.01003692,0.01630972,0.009545352,0.003906394,0.005016615,0.2705071],"study_design_scores_gemma":[0.0003570288,0.005297469,0.8051078,0.000691737,0.0006869943,0.002119887,0.008599784,0.1220527,0.02467686,0.007700024,0.02238328,0.0003263526],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9776281,0.000734204,0.01563579,0.0004363793,0.0000476187,0.0001015012,0.0002485509,0.00110068,0.004067041],"genre_scores_gemma":[0.987564,0.0001745324,0.01091961,0.0000443931,0.00001931819,0.00004440334,0.0004155512,0.0002782744,0.0005397863],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04672338,"threshold_uncertainty_score":0.2470998,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1816990682329545,"score_gpt":0.3961308599928329,"score_spread":0.2144317917598785,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}