{"id":"W4404781502","doi":"10.18653/v1/2024.wmt-1.4","title":"Findings of the WMT 2024 Shared Task of the Open Language Data Initiative","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Task (project management); Open data; Natural language processing; Artificial intelligence; World Wide Web; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1923681,0.002211588,0.002836123,0.008266991,0.008994604,0.01228624,0.005193749,0.005377011,0.01290352],"category_scores_gemma":[0.4065212,0.001122523,0.002316136,0.01006571,0.006508264,0.009779027,0.03246406,0.005786396,0.01161523],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007322373,"about_ca_system_score_gemma":0.01850573,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04439199,"about_ca_topic_score_gemma":0.05144248,"domain_scores_codex":[0.7895513,0.1224061,0.01074301,0.01308903,0.05574883,0.00846175],"domain_scores_gemma":[0.5153773,0.213608,0.02225723,0.05937359,0.1394226,0.04996134],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002229308,0.001131695,0.01711004,0.002525672,0.0003489933,0.000623077,0.00724147,0.0013939,0.002104235,0.006921589,0.9308392,0.0275308],"study_design_scores_gemma":[0.001504285,0.0008847743,0.1313899,0.00206816,0.0004430426,0.0009580482,0.02056787,0.004507269,0.005364373,0.01846436,0.8132006,0.0006473369],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.2302911,0.007759918,0.0221854,0.1378773,0.01509143,0.0068179,0.5026994,0.005107357,0.07217015],"genre_scores_gemma":[0.2319768,0.001856807,0.0406808,0.02459574,0.005969182,0.01529643,0.6365831,0.008391565,0.03464956],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1923681,"threshold_uncertainty_score":0.9959539,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04645345507253468,"score_gpt":0.3402527732463607,"score_spread":0.293799318173826,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}