{"meta":{"query_hash":"96b9e3b2fc0b","filters":{"venue":"Natural language processing."},"cohort_total":3,"direct_labels_cover":1,"predictions_cover":3,"exported":3,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/96b9e3b2fc0b","api":"https://metacan.xera.ac/api/v1/cohort?venue=Natural+language+processing."},"results":[{"id":"W4397012280","doi":"10.1017/nlp.2024.7","title":"A survey of context in neural machine translation and its evaluation","year":2024,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"Dublin City University; Science Foundation Ireland","keywords":"Machine translation; Computer science; Artificial intelligence; Evaluation of machine translation; Context (archaeology); Natural language processing; Terminology; Machine translation software usability; Consistency (knowledge bases); Example-based machine translation; Sentence; Computer-assisted translation; Task (project management); Paragraph; Transfer-based machine translation; Machine learning; Linguistics; World Wide Web; Engineering","score_opus":0.030556959899479947,"score_gpt":0.33267595850217146,"score_spread":0.30211899860269154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4397012280","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018729841,0.9748766,0.0051524816,0.0003949656,0.0001534186,0.00028896917,0.000011097464,0.00031814136,0.0000745157],"genre_scores_gemma":[0.9890822,0.000033692082,0.010660719,0.00009984415,0.000021757201,0.000017447202,0.000036154877,0.00001410868,0.000034058812],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984742,0.0001774378,0.00031038918,0.00041062458,0.00042651457,0.00020087125],"domain_scores_gemma":[0.99932677,0.00016326664,0.000087074026,0.00015497794,0.00022927656,0.00003864651],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012004287,0.00016747238,0.00020802482,0.00032167177,0.00004143562,0.00020812947,0.00037131077,0.00010437493,0.000008807088],"category_scores_gemma":[0.00031167569,0.00013363696,0.00003076506,0.0010129871,0.000034593653,0.0011165764,0.00007800641,0.00037403195,0.0000016238407],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000014144861,0.000013474613,0.00026052957,0.00024038299,0.000004058089,0.00001853542,0.0027351785,0.0000032528635,0.0012153733,0.00008019738,0.0000101163405,0.9954048],"study_design_scores_gemma":[0.0002483814,0.000030793894,0.0027761064,0.00025856943,0.000009357004,0.000027434304,0.000033197506,0.99489444,0.0014615408,0.00009091802,0.000011904878,0.00015735414],"about_ca_topic_score_codex":0.00037443786,"about_ca_topic_score_gemma":0.00052349304,"teacher_disagreement_score":0.9952474,"about_ca_system_score_codex":0.00005292219,"about_ca_system_score_gemma":0.00012722985,"threshold_uncertainty_score":0.5449557},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"review","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W4399298003","doi":"10.1017/nlp.2024.5","title":"Calibration and context in human evaluation of machine translation","year":2024,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Calibration; Context (archaeology); Translation (biology); Machine translation; Computer science; Artificial intelligence; Natural language processing; Machine learning; Chemistry; Biology; Mathematics; Statistics; Biochemistry","score_opus":0.019952542382340694,"score_gpt":0.32952392468542374,"score_spread":0.30957138230308306,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399298003","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028647747,0.908371,0.06125342,0.0006087626,0.000115346535,0.00031769008,0.000003835963,0.00045605833,0.00022616058],"genre_scores_gemma":[0.96854496,0.000015553145,0.03125207,0.000073718125,0.000032872485,0.00001757247,0.000022742908,0.000010967192,0.000029525409],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987186,0.00009431426,0.00027042016,0.00032851484,0.00044744197,0.00014071562],"domain_scores_gemma":[0.99957865,0.000049929553,0.0000750336,0.00014958838,0.000119819735,0.000026975866],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00080579944,0.00012934786,0.0001484569,0.00030466146,0.000052144238,0.00021910951,0.00025023785,0.0000939287,0.000009662463],"category_scores_gemma":[0.0000740608,0.00010555539,0.000028532077,0.00060077274,0.00004534278,0.0012474025,0.0000494023,0.00027037173,4.1606836e-7],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000039811266,0.000012470769,0.00005375931,0.00019792187,0.000003229616,0.000008534567,0.0040622517,0.0000034072314,0.016977135,0.0019292772,0.000012449569,0.9767356],"study_design_scores_gemma":[0.00026583404,0.00002622875,0.000104928986,0.00033207325,0.000015940377,0.000019133335,0.00009079301,0.9762081,0.019078305,0.003697028,0.000025132349,0.00013648259],"about_ca_topic_score_codex":0.00013785031,"about_ca_topic_score_gemma":0.000156289,"teacher_disagreement_score":0.9765991,"about_ca_system_score_codex":0.00006198776,"about_ca_system_score_gemma":0.000095100266,"threshold_uncertainty_score":0.4304424},"labels":[],"label_agreement":null},{"id":"W4411558583","doi":"10.1017/nlp.2024.6","title":"Editors’ foreword","year":2025,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Psychology; Computer science; Cognitive science","score_opus":0.004900954798458003,"score_gpt":0.2795578525872755,"score_spread":0.2746568977888175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411558583","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00077744474,0.42422137,0.5492843,0.0059556626,0.004222487,0.00041405758,0.000003424545,0.0047249063,0.010396371],"genre_scores_gemma":[0.71540165,0.0000073227957,0.27874196,0.0018684278,0.00059362,0.000028903502,0.000007305089,0.000013652042,0.003337174],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981258,0.00004560964,0.0003064973,0.0006171752,0.00040566965,0.0004992858],"domain_scores_gemma":[0.9988218,0.0000804366,0.0001381735,0.00063503784,0.00024687045,0.00007765907],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00028157036,0.00028587098,0.00026341502,0.00036409963,0.0002742897,0.0005418306,0.0019323645,0.0001965005,0.000014165971],"category_scores_gemma":[0.00031371918,0.0002277929,0.00010202362,0.0014297973,0.00008613151,0.0011741226,0.00053306675,0.0006373686,0.000024747724],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009841854,0.000033013122,0.00006970305,0.00012421548,0.000011254152,0.00004568655,0.0006126185,3.730497e-7,0.002105688,0.0079465,0.016893491,0.97214764],"study_design_scores_gemma":[0.0036014875,0.00028757087,0.0010527124,0.0031031056,0.00016374432,0.0003137317,0.0010352248,0.4041208,0.27112758,0.110754825,0.19998811,0.0044511114],"about_ca_topic_score_codex":0.000032059943,"about_ca_topic_score_gemma":0.000010527232,"teacher_disagreement_score":0.9676965,"about_ca_system_score_codex":0.000113820715,"about_ca_system_score_gemma":0.00022396266,"threshold_uncertainty_score":0.92891246},"labels":[],"label_agreement":null}]}