{"id":"W4392083945","doi":"10.18653/v1/2024.emnlp-main.1033","title":"Enhanced Hallucination Detection in Neural Machine Translation through Simple Detector Aggregation","year":2024,"lang":"en","type":"article","venue":"","topic":"Cell Image Analysis Techniques","field":"Biochemistry, Genetics and Molecular Biology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Hallucinating; Detector; Computer science; Machine translation; Translation (biology); Software deployment; Artificial intelligence; Simple (philosophy); Machine learning; Rank (graph theory); Natural language processing; Telecommunications; Mathematics; Software engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001251116,0.0001023781,0.00007782281,0.0000951255,0.0000285015,0.00004411441,0.00005514297,0.0001036722,0.00005454515],"category_scores_gemma":[0.00002910759,0.00009854192,0.00006744962,0.0002431431,0.00001518786,0.00002323674,0.00001453937,0.00007654401,0.000007265834],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003030355,"about_ca_system_score_gemma":0.00001220454,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002065086,"about_ca_topic_score_gemma":0.003118325,"domain_scores_codex":[0.9992643,0.00004664389,0.0001892218,0.000295939,0.00009098444,0.000112921],"domain_scores_gemma":[0.9997551,0.0000111792,0.00002744873,0.0001548461,0.00003638612,0.00001501547],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00001433553,0.00001031294,0.00002156392,0.00001136737,0.000008030924,0.000001003551,0.00003976594,0.00008965719,0.6651568,0.000004361787,0.0000254841,0.3346173],"study_design_scores_gemma":[0.0001271619,0.00008593943,0.0002356462,0.000006837888,0.000013913,0.000003030898,0.00001035348,0.1288871,0.8678555,0.0001397857,0.002533625,0.0001011011],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4282311,0.001862427,0.567389,0.00006983312,0.00003483607,0.0001947817,0.000001551975,0.0001058914,0.00211066],"genre_scores_gemma":[0.9977044,0.0001666427,0.001535371,0.00006820311,0.0000801745,0.00003485927,0.0001861947,0.00001750333,0.0002066886],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5694733,"threshold_uncertainty_score":0.4018423,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0093627393100804,"score_gpt":0.2806969907406583,"score_spread":0.2713342514305779,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}