{"id":"W6947949615","doi":"10.48448/wbq8-xt62","title":"AraBench: Benchmarking Dialectal Arabic-English Machine Translation","year":2020,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Research Data Management Practices","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Unavailability; Suite; Machine translation; Benchmarking; Variation (astronomy); Arabic; Standard English","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006603252,0.00345537,0.00132243,0.004018421,0.001743658,0.002806002,0.003434994,0.001796951,0.01324673],"category_scores_gemma":[0.01664876,0.00063188,0.001163871,0.004115527,0.001208062,0.0032072,0.004489768,0.002036357,0.01984101],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001273353,"about_ca_system_score_gemma":0.001827185,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009012235,"about_ca_topic_score_gemma":0.01235545,"domain_scores_codex":[0.9913709,0.003797112,0.001042758,0.001737488,0.001538647,0.0005132083],"domain_scores_gemma":[0.9917265,0.002761058,0.0002614312,0.002284492,0.002440596,0.0005259121],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003074598,0.001905155,0.00818321,0.004759939,0.001237672,0.000844524,0.001185282,0.03575179,0.02232891,0.00468542,0.4281746,0.4878689],"study_design_scores_gemma":[0.002101225,0.003461712,0.02563962,0.0009351868,0.000681356,0.002335229,0.002673212,0.3739015,0.1212436,0.01655908,0.4499338,0.0005345343],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.3551734,0.01871269,0.1540985,0.002168605,0.004987082,0.002618846,0.109761,0.2712102,0.08126967],"genre_scores_gemma":[0.3335311,0.002034474,0.1533804,0.00110676,0.0003304998,0.001502292,0.4745071,0.01169648,0.02191088],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01324673,"threshold_uncertainty_score":0.0443148,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07131721733068196,"score_gpt":0.3264935714448415,"score_spread":0.2551763541141596,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}