{"id":"W2121771196","doi":"","title":"AMBER: A Modified BLEU, Enhanced Ranking Metric","year":2011,"lang":"en","type":"article","venue":"NPARC","topic":"Topic Modeling","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"BLEU; Metric (unit); Computer science; Machine translation; Ranking (information retrieval); Artificial intelligence; Natural language processing; Sentence; Consistency (knowledge bases); Task (project management); Recall; Translation (biology); Speech recognition; Linguistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007470604,0.001814935,0.001932926,0.007155542,0.001144719,0.001948298,0.002393809,0.001483652,0.004626418],"category_scores_gemma":[0.02488495,0.0003934355,0.0007424435,0.004569463,0.0005889941,0.003275244,0.001679909,0.001634907,0.004494233],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001164854,"about_ca_system_score_gemma":0.001611877,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002935775,"about_ca_topic_score_gemma":0.006459615,"domain_scores_codex":[0.9864295,0.0075099,0.0008683563,0.001199096,0.003649336,0.000343834],"domain_scores_gemma":[0.9862213,0.005586582,0.0009782817,0.00287044,0.004062296,0.0002811219],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008435055,0.000388785,0.004931403,0.001131024,0.0006479234,0.0001238167,0.0002806773,0.03818749,0.01790112,0.01252466,0.07609399,0.8469456],"study_design_scores_gemma":[0.0005967579,0.003471964,0.01962442,0.0003637505,0.000860724,0.001896834,0.0002225753,0.6729482,0.08248059,0.04863255,0.1679524,0.0009490927],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05065418,0.009815158,0.8735564,0.000824831,0.001159674,0.0008835369,0.01070344,0.0241111,0.02829165],"genre_scores_gemma":[0.3145091,0.001654362,0.6406832,0.000697226,0.0006220406,0.001585042,0.02203641,0.003613096,0.01459943],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007470604,"threshold_uncertainty_score":0.03950882,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04939886284340537,"score_gpt":0.2435130355547379,"score_spread":0.1941141727113325,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}