{"meta":{"query_hash":"96b9e3b2fc0b","filters":{"venue":"Natural language processing."},"cohort_total":3,"direct_labels_cover":1,"predictions_cover":3,"exported":3,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/96b9e3b2fc0b","api":"https://metacan.xera.ac/api/v1/cohort?venue=Natural+language+processing."},"results":[{"id":"W4397012280","doi":"10.1017/nlp.2024.7","title":"A survey of context in neural machine translation and its evaluation","year":2024,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"Dublin City University; Science Foundation Ireland","keywords":"Machine translation; Computer science; Artificial intelligence; Evaluation of machine translation; Context (archaeology); Natural language processing; Terminology; Machine translation software usability; Consistency (knowledge bases); Example-based machine translation; Sentence; Computer-assisted translation; Task (project management); Paragraph; Transfer-based machine translation; Machine learning; Linguistics; World Wide Web; Engineering","score_opus":0.030556959899479947,"score_gpt":0.33267595850217146,"score_spread":0.30211899860269154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4397012280","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17011558,0.66231364,0.12464162,0.0035201954,0.00090769015,0.0005511797,0.0013558522,0.0029700855,0.03362423],"genre_scores_gemma":[0.8122588,0.06769282,0.11174368,0.0010037926,0.000789605,0.00043112386,0.0026380115,0.0006064857,0.0028356737],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98102933,0.012083333,0.0016058793,0.0015827799,0.0033802798,0.00031839617],"domain_scores_gemma":[0.97104836,0.019442126,0.0013356643,0.0017467352,0.006013337,0.00041394992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0145920655,0.0015053155,0.0022158772,0.0054267994,0.0009528288,0.0029869883,0.0023208803,0.002165894,0.0024535498],"category_scores_gemma":[0.042875595,0.0005654492,0.0010717568,0.0049382616,0.0010596183,0.0033049393,0.0019217391,0.0012976509,0.00065929943],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016117473,0.00022742429,0.008395502,0.0038974353,0.00092337007,0.00009182491,0.0002090971,0.043230996,0.0029817754,0.005636111,0.0055218353,0.9272729],"study_design_scores_gemma":[0.0005812291,0.004859946,0.03096424,0.00749666,0.002693787,0.0010686708,0.0010997849,0.7670239,0.04423236,0.05328752,0.08629181,0.0004000997],"about_ca_topic_score_codex":0.0064816517,"about_ca_topic_score_gemma":0.0059772097,"teacher_disagreement_score":0.0145920655,"about_ca_system_score_codex":0.002650352,"about_ca_system_score_gemma":0.0015265252,"threshold_uncertainty_score":0.07717115},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"review","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W4399298003","doi":"10.1017/nlp.2024.5","title":"Calibration and context in human evaluation of machine translation","year":2024,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Calibration; Context (archaeology); Translation (biology); Machine translation; Computer science; Artificial intelligence; Natural language processing; Machine learning; Chemistry; Biology; Mathematics; Statistics; Biochemistry","score_opus":0.019952542382340694,"score_gpt":0.32952392468542374,"score_spread":0.30957138230308306,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399298003","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6657237,0.0113585945,0.27951953,0.0024298774,0.000580689,0.0014904718,0.00047263247,0.0013853206,0.037039302],"genre_scores_gemma":[0.9514693,0.00030714297,0.046351895,0.00034733795,0.000120580175,0.0004972448,0.00015809642,0.0001955351,0.000552889],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.6229384,0.33409697,0.00805994,0.013059717,0.020592835,0.0012521766],"domain_scores_gemma":[0.602674,0.29898867,0.03130128,0.02853955,0.035820194,0.0026762756],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.12338182,0.0012773442,0.0011317801,0.0031557623,0.0023810861,0.0051637734,0.0021042337,0.0021479414,0.0020985294],"category_scores_gemma":[0.38905817,0.0009923097,0.0006853218,0.0023246363,0.004251792,0.004760565,0.00757032,0.0022991695,0.00062380225],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013265053,0.0010830809,0.121547,0.003837435,0.0027583987,0.00070531946,0.035930835,0.048922397,0.06267169,0.03181332,0.007383508,0.670082],"study_design_scores_gemma":[0.0015032112,0.008361804,0.29770315,0.005963982,0.002537868,0.002746942,0.013999833,0.2096984,0.17562464,0.21777008,0.062047374,0.0020426463],"about_ca_topic_score_codex":0.0013799558,"about_ca_topic_score_gemma":0.0020224468,"teacher_disagreement_score":0.87661815,"about_ca_system_score_codex":0.002286055,"about_ca_system_score_gemma":0.001582256,"threshold_uncertainty_score":0.65251327},"labels":[],"label_agreement":null},{"id":"W4411558583","doi":"10.1017/nlp.2024.6","title":"Editors’ foreword","year":2025,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Psychology; Computer science; Cognitive science","score_opus":0.004900954798458003,"score_gpt":0.2795578525872755,"score_spread":0.2746568977888175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411558583","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000118496886,0.005257195,0.0004327724,0.046707407,0.9267867,0.00005229886,0.00048488277,0.00020460313,0.019955518],"genre_scores_gemma":[0.0021541708,0.009427588,0.00064561766,0.049242195,0.77226096,0.00013148141,0.0008199013,0.00042809398,0.16489],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99771786,0.00028266927,0.00026210462,0.00045704184,0.0010613465,0.00021894669],"domain_scores_gemma":[0.9877302,0.0026601767,0.0007120003,0.00048654372,0.0064814864,0.001929603],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028243286,0.0015565968,0.001462624,0.0024137353,0.0014563841,0.006337774,0.0021898106,0.0042922595,0.15813571],"category_scores_gemma":[0.01572473,0.00047741132,0.001162962,0.0014528637,0.0007724263,0.003780216,0.0018210055,0.006871327,0.12448617],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012191943,0.0000057525285,0.000019307083,0.000074203,0.000002310415,0.000018226781,0.0000047683498,0.000017231197,0.000038818147,0.00026242048,0.9894014,0.010143355],"study_design_scores_gemma":[0.000010122971,0.000016221153,0.0001405164,0.00016311693,0.0000049027503,0.000069667214,0.000014683003,0.000050912735,0.00010332523,0.00046641155,0.998953,0.000007099789],"about_ca_topic_score_codex":0.000908301,"about_ca_topic_score_gemma":0.0016978274,"teacher_disagreement_score":0.15813571,"about_ca_system_score_codex":0.0014495213,"about_ca_system_score_gemma":0.0019168485,"threshold_uncertainty_score":0.5290167},"labels":[],"label_agreement":null}]}