{"id":"W7106650730","doi":"10.32913/mic-ict-research.v2025.n1.1277","title":"A Rich High-Order Mutation Testing Dataset for Software Fortification","year":2024,"lang":"","type":"article","venue":"Research and Development on Information and Communication Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Mutation; Mutation testing; Random forest; Software; Software testing; Test (biology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001216813,0.001047957,0.0005243696,0.002510353,0.0008091828,0.0007597227,0.002199398,0.001984756,0.001367648],"category_scores_gemma":[0.006269273,0.0002714189,0.0008557221,0.002428184,0.0005965206,0.000743907,0.0009736361,0.001618108,0.0009250327],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009314811,"about_ca_system_score_gemma":0.001328882,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008647153,"about_ca_topic_score_gemma":0.02074276,"domain_scores_codex":[0.9983961,0.0002464854,0.0001603983,0.0003787552,0.0006863743,0.0001318287],"domain_scores_gemma":[0.9945273,0.002202896,0.0006684033,0.0009846895,0.001270622,0.0003461528],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001460958,0.003375338,0.1617095,0.003340214,0.0006572278,0.005241799,0.001039568,0.1273522,0.03116322,0.007236967,0.4136482,0.2437749],"study_design_scores_gemma":[0.0008271224,0.001331142,0.3259947,0.0005563315,0.0002806458,0.00500951,0.001033777,0.2641154,0.05403697,0.01381882,0.3326749,0.0003207405],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.624841,0.002788927,0.02988227,0.001833355,0.0003696319,0.0006545252,0.3227244,0.0074311,0.009474747],"genre_scores_gemma":[0.4026549,0.0006189396,0.05278505,0.0006191377,0.0001121684,0.0007274001,0.5387415,0.0004763516,0.003264616],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008647153,"threshold_uncertainty_score":0.01719362,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1138390251804364,"score_gpt":0.3782881379797234,"score_spread":0.264449112799287,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}