{"id":"W4403887388","doi":"10.1007/s10664-024-10549-2","title":"Towards effectively testing machine translation systems from white-box perspectives","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Concordia University; University of Waterloo","funders":"","keywords":"Computer science; Translation (biology); Machine translation; White box; Systems engineering; Artificial intelligence; Engineering; Software engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002389933,0.0002412282,0.0002225274,0.0001863072,0.00007331328,0.0005232217,0.0004800062,0.0001241119,0.000005911213],"category_scores_gemma":[0.0006948421,0.0002076635,0.00008158614,0.0008999271,0.00001606123,0.0006453259,0.000118811,0.0004470893,0.00001601863],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002335821,"about_ca_system_score_gemma":0.0000637554,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001392397,"about_ca_topic_score_gemma":0.000001216379,"domain_scores_codex":[0.998576,0.00004401064,0.0002177907,0.0005558447,0.0003171601,0.0002891883],"domain_scores_gemma":[0.9987878,0.0007505735,0.00002963465,0.0002559671,0.00007753926,0.00009849368],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002576158,0.0001845899,0.02374866,0.001841271,0.0004003766,0.0009334626,0.02710804,0.02066796,0.01732184,0.008618684,0.001516927,0.8976324],"study_design_scores_gemma":[0.0001488986,0.00009628874,0.005722585,0.000764186,0.00002502429,0.00005327455,0.00002861586,0.9868129,0.002154579,0.001851704,0.001759625,0.0005823542],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004703367,0.04816967,0.939902,0.0003420204,0.000503942,0.0001748607,0.00001367687,0.006141673,0.00004878625],"genre_scores_gemma":[0.5284211,0.000003910083,0.4713141,0.00001727171,0.0001752995,0.00002428042,0.000004872275,0.00002476619,0.00001443104],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9661449,"threshold_uncertainty_score":0.846827,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02217214222265518,"score_gpt":0.2788882783243808,"score_spread":0.2567161361017257,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}