{"id":"W4406074029","doi":"10.48550/arxiv.2407.16470","title":"Machine Translation Hallucination Detection for Low and High Resource Languages using Large Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Text Readability and Simplification","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Computer science; Machine translation; Translation (biology); Natural language processing; Artificial intelligence; Resource (disambiguation); Linguistics; Psychology; Chemistry; Philosophy","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002771216,0.001320208,0.000923837,0.001284541,0.0006694573,0.001987381,0.0008319562,0.0009331839,0.002601531],"category_scores_gemma":[0.01124877,0.0003511026,0.0009710181,0.001180771,0.0006715556,0.002993929,0.001723443,0.001485111,0.002758921],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006275448,"about_ca_system_score_gemma":0.0009717404,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003947814,"about_ca_topic_score_gemma":0.007104561,"domain_scores_codex":[0.998064,0.00101454,0.0001366698,0.0003665891,0.0002900655,0.0001280872],"domain_scores_gemma":[0.9957001,0.002198537,0.0003154114,0.0009282391,0.0006605765,0.0001972157],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00129574,0.0004493261,0.01571925,0.0007662416,0.0006226281,0.001110691,0.0008021938,0.1433593,0.0402347,0.006390348,0.02128536,0.7679642],"study_design_scores_gemma":[0.00005009246,0.0002608181,0.003184288,0.0000546036,0.00008076841,0.0004729186,0.0004491005,0.9543313,0.02651114,0.008760671,0.005792242,0.00005211062],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4731406,0.003537414,0.4920852,0.001655898,0.0004696264,0.0001629338,0.001273108,0.01835234,0.00932288],"genre_scores_gemma":[0.8900662,0.0005504378,0.1017534,0.0002868648,0.0001139414,0.00005970693,0.003345127,0.0005609479,0.003263342],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003947814,"threshold_uncertainty_score":0.01465577,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05613422021312714,"score_gpt":0.2141775388453185,"score_spread":0.1580433186321914,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}