{"id":"W4385977849","doi":"10.1007/s41870-023-01390-9","title":"A study of the interactive role of metamorphic testing and machine learning in the quality assurance of a deep learning forecasting application","year":2023,"lang":"en","type":"article","venue":"International Journal of Information Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"Mitacs","keywords":"Machine learning; Computer science; Artificial intelligence; Metamorphic rock; Generalization; Software quality; Deep learning; Quality (philosophy); Software; Algorithm; Software development; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001340369,0.0002274793,0.0001882986,0.0004301264,0.0003371501,0.0009142159,0.0006624373,0.0005694526,0.001573253],"category_scores_gemma":[0.01491049,0.0001389946,0.0001512953,0.0003273935,0.000620454,0.001124031,0.0004224565,0.0007624025,0.00008526118],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007005906,"about_ca_system_score_gemma":0.0006052196,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003402596,"about_ca_topic_score_gemma":0.002279548,"domain_scores_codex":[0.9993336,0.0002754163,0.00002605013,0.00008979084,0.0001918895,0.00008315525],"domain_scores_gemma":[0.9834,0.01300122,0.001263279,0.0008175215,0.001091782,0.000426179],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002405529,0.002921476,0.1825963,0.0004406991,0.0002306686,0.003362109,0.004629752,0.3181546,0.1336793,0.06107493,0.003171589,0.2873331],"study_design_scores_gemma":[0.00002398465,0.0005022683,0.0202932,0.00002010089,0.00003155075,0.0001827738,0.00040344,0.958448,0.01483695,0.004384491,0.0008539164,0.0000193037],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9804404,0.0001126415,0.01635825,0.0002181014,0.000008997948,0.00001908153,0.0000129225,0.0001041226,0.002725486],"genre_scores_gemma":[0.9982725,0.00001214879,0.001439489,0.000009055884,0.000002069741,0.000001742955,0.000005315263,0.000007510929,0.0002501038],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003402596,"threshold_uncertainty_score":0.007088661,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03081026342351223,"score_gpt":0.3040353657264644,"score_spread":0.2732251023029522,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}