{"id":"W4403887388","doi":"10.1007/s10664-024-10549-2","title":"Towards effectively testing machine translation systems from white-box perspectives","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Concordia University; University of Waterloo","funders":"","keywords":"Computer science; Translation (biology); Machine translation; White box; Systems engineering; Artificial intelligence; Engineering; Software engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0172734,0.001712367,0.0012219,0.002380888,0.001031276,0.005320158,0.00302456,0.003789167,0.005006432],"category_scores_gemma":[0.1130148,0.0012648,0.0009909312,0.001401962,0.003712038,0.01301665,0.003983439,0.003402342,0.001239424],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001103138,"about_ca_system_score_gemma":0.003309419,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001810842,"about_ca_topic_score_gemma":0.002813133,"domain_scores_codex":[0.9697939,0.02109341,0.001350647,0.002358413,0.00448338,0.0009203235],"domain_scores_gemma":[0.8343731,0.1376906,0.003751718,0.0145667,0.008431872,0.001186088],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001367015,0.002335174,0.02211641,0.001809748,0.0004873904,0.00216642,0.004273767,0.1245161,0.09882133,0.2421831,0.01036766,0.489556],"study_design_scores_gemma":[0.0001723348,0.0004157578,0.001874954,0.0002038065,0.0001194834,0.0004472224,0.0009892301,0.6873204,0.0551329,0.2479062,0.005359741,0.0000580611],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09494362,0.0002843144,0.8935737,0.001760985,0.00008449572,0.0001854717,0.0002176639,0.005225143,0.003724645],"genre_scores_gemma":[0.5383083,0.0001573749,0.457419,0.0006555539,0.00009743655,0.0001955053,0.0005482087,0.001089346,0.001529349],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0172734,"threshold_uncertainty_score":0.09135157,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02217214222265518,"score_gpt":0.2788882783243808,"score_spread":0.2567161361017257,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}