{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":3,"total_is_capped":false,"direct_labels_cover":1,"predictions_cover":3,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"96b9e3b2fc0b","filters":{"venue":"Natural language processing."}},"results":[{"id":"W4397012280","doi":"10.1017/nlp.2024.7","title":"A survey of context in neural machine translation and its evaluation","year":2024,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"National Research Council Canada","funders":"Dublin City University; Science Foundation Ireland","keywords":"Machine translation; Computer science; Artificial intelligence; Evaluation of machine translation; Context (archaeology); Natural language processing; Terminology; Machine translation software usability; Consistency (knowledge bases); Example-based machine translation; Sentence; Computer-assisted translation; Task (project management); Paragraph; Transfer-based machine translation; Machine learning; Linguistics; World Wide Web; Engineering","authors":[{"name":"Sheila Castilho","is_ca":false},{"name":"Rebecca Knowles","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03055695989947995,"gpt":0.3326759585021715,"spread":0.3021189986026915,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001200429,0.0001674724,0.0002080248,0.0003216718,0.00004143562,0.0002081295,0.0003713108,0.0001043749,0.000008807088],"category_scores_gemma":[0.0003116757,0.000133637,0.00003076506,0.001012987,0.00003459365,0.001116576,0.00007800641,0.000374032,0.000001623841],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005292219,"about_ca_system_score_gemma":0.0001272299,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003744379,"about_ca_topic_score_gemma":0.000523493,"domain_scores_codex":[0.9984742,0.0001774378,0.0003103892,0.0004106246,0.0004265146,0.0002008712],"domain_scores_gemma":[0.9993268,0.0001632666,0.00008707403,0.0001549779,0.0002292766,0.00003864651],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001414486,0.00001347461,0.0002605296,0.000240383,0.000004058089,0.00001853542,0.002735178,0.000003252864,0.001215373,0.00008019738,0.00001011634,0.9954048],"study_design_scores_gemma":[0.0002483814,0.00003079389,0.002776106,0.0002585694,0.000009357004,0.0000274343,0.00003319751,0.9948944,0.001461541,0.00009091802,0.00001190488,0.0001573541],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.01872984,0.9748766,0.005152482,0.0003949656,0.0001534186,0.0002889692,0.00001109746,0.0003181414,0.0000745157],"genre_scores_gemma":[0.9890822,0.00003369208,0.01066072,0.00009984415,0.0000217572,0.0000174472,0.00003615488,0.00001410868,0.00003405881],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9952474,"threshold_uncertainty_score":0.5449557,"prediction_status":"machine_predicted_unvalidated"},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"review","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W4399298003","doi":"10.1017/nlp.2024.5","title":"Calibration and context in human evaluation of machine translation","year":2024,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"National Research Council Canada","funders":"","keywords":"Calibration; Context (archaeology); Translation (biology); Machine translation; Computer science; Artificial intelligence; Natural language processing; Machine learning; Chemistry; Biology; Mathematics; Statistics; Biochemistry","authors":[{"name":"Rebecca Knowles","is_ca":true},{"name":"Chi-kiu Lo","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01995254238234069,"gpt":0.3295239246854237,"spread":0.3095713823030831,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008057994,0.0001293479,0.0001484569,0.0003046615,0.00005214424,0.0002191095,0.0002502378,0.0000939287,0.000009662463],"category_scores_gemma":[0.0000740608,0.0001055554,0.00002853208,0.0006007727,0.00004534278,0.001247403,0.0000494023,0.0002703717,4.160684e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006198776,"about_ca_system_score_gemma":0.00009510027,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001378503,"about_ca_topic_score_gemma":0.000156289,"domain_scores_codex":[0.9987186,0.00009431426,0.0002704202,0.0003285148,0.000447442,0.0001407156],"domain_scores_gemma":[0.9995787,0.00004992955,0.0000750336,0.0001495884,0.0001198197,0.00002697587],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003981127,0.00001247077,0.00005375931,0.0001979219,0.000003229616,0.000008534567,0.004062252,0.000003407231,0.01697714,0.001929277,0.00001244957,0.9767356],"study_design_scores_gemma":[0.000265834,0.00002622875,0.000104929,0.0003320733,0.00001594038,0.00001913333,0.00009079301,0.9762081,0.0190783,0.003697028,0.00002513235,0.0001364826],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.02864775,0.908371,0.06125342,0.0006087626,0.0001153465,0.0003176901,0.000003835963,0.0004560583,0.0002261606],"genre_scores_gemma":[0.968545,0.00001555315,0.03125207,0.00007371813,0.00003287249,0.00001757247,0.00002274291,0.00001096719,0.00002952541],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9765991,"threshold_uncertainty_score":0.4304424,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4411558583","doi":"10.1017/nlp.2024.6","title":"Editors’ foreword","year":2025,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"National Research Council Canada","funders":"","keywords":"Psychology; Computer science; Cognitive science","authors":[{"name":"Sheila Castilho","is_ca":false},{"name":"Rebecca Knowles","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.004900954798458003,"gpt":0.2795578525872755,"spread":0.2746568977888175,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002815704,0.000285871,0.000263415,0.0003640996,0.0002742897,0.0005418306,0.001932365,0.0001965005,0.00001416597],"category_scores_gemma":[0.0003137192,0.0002277929,0.0001020236,0.001429797,0.00008613151,0.001174123,0.0005330667,0.0006373686,0.00002474772],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001138207,"about_ca_system_score_gemma":0.0002239627,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003205994,"about_ca_topic_score_gemma":0.00001052723,"domain_scores_codex":[0.9981258,0.00004560964,0.0003064973,0.0006171752,0.0004056696,0.0004992858],"domain_scores_gemma":[0.9988218,0.0000804366,0.0001381735,0.0006350378,0.0002468704,0.00007765907],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000009841854,0.00003301312,0.00006970305,0.0001242155,0.00001125415,0.00004568655,0.0006126185,3.730497e-7,0.002105688,0.0079465,0.01689349,0.9721476],"study_design_scores_gemma":[0.003601487,0.0002875709,0.001052712,0.003103106,0.0001637443,0.0003137317,0.001035225,0.4041208,0.2711276,0.1107548,0.1999881,0.004451111],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0007774447,0.4242214,0.5492843,0.005955663,0.004222487,0.0004140576,0.000003424545,0.004724906,0.01039637],"genre_scores_gemma":[0.7154016,0.000007322796,0.278742,0.001868428,0.00059362,0.0000289035,0.000007305089,0.00001365204,0.003337174],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9676965,"threshold_uncertainty_score":0.9289125,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}