{"id":"W2973276266","doi":"10.48550/arxiv.1909.09268","title":"Towards Neural Language Evaluators","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Metric (unit); Natural language processing; Transformer; Artificial intelligence; BLEU; Machine learning; Information retrieval; Machine translation; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0002801797,0.0002785608,0.000298977,0.0002064626,0.00006886589,0.0001257418,0.002478969,0.0002526959,0.0000550636],"category_scores_gemma":[0.00003275197,0.0003146511,0.0002183413,0.0003051449,0.00004013156,0.0003161028,0.003068824,0.0005889575,0.0002403747],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00020215,"about_ca_system_score_gemma":0.0002768814,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003145168,"about_ca_topic_score_gemma":0.00001743372,"domain_scores_codex":[0.9980412,0.0001392381,0.0001671195,0.001146632,0.0001478264,0.0003579454],"domain_scores_gemma":[0.9976305,0.00004433037,0.0001623192,0.001925845,0.00009831653,0.0001386684],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001063423,0.00004666011,0.00278094,0.0001033109,0.00007421684,0.0004369185,0.001154225,0.8711441,0.00004388966,0.1161155,0.0003025054,0.007787073],"study_design_scores_gemma":[0.0002758088,0.00002173904,0.0007389361,0.00003686032,0.00003202024,0.000004169196,0.00007918988,0.9892182,0.00008375121,0.008963221,0.0001932173,0.0003528458],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4742303,0.00008638452,0.519201,0.0001035151,0.001171713,0.0001966665,0.00000502318,0.0002485773,0.004756812],"genre_scores_gemma":[0.994298,0.00002352884,0.00287545,0.0001763188,0.000115414,4.992042e-7,0.000007055266,0.00001661489,0.002487099],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5200677,"threshold_uncertainty_score":0.9999306,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08674456871558647,"score_gpt":0.211965579487791,"score_spread":0.1252210107722045,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}