{"meta":{"query_hash":"7aa4c1c33ff6","filters":{"venue":"Journal of Natural Language Processing"},"cohort_total":4,"direct_labels_cover":0,"predictions_cover":4,"exported":4,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/7aa4c1c33ff6","api":"https://metacan.xera.ac/api/v1/cohort?venue=Journal+of+Natural+Language+Processing"},"results":[{"id":"W2062246499","doi":"10.5715/jnlp.18.217","title":"Construction of Context Models for Word Sense Disambiguation","year":2011,"lang":"en","type":"article","venue":"Journal of Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Ministry of Education, Culture, Sports, Science and Technology","keywords":"Word-sense disambiguation; Word (group theory); Context (archaeology); Computer science; SemEval; Natural language processing; Linguistics; Artificial intelligence; History; Philosophy; Engineering","score_opus":0.02621211769329632,"score_gpt":0.2813093315516963,"score_spread":0.25509721385839995,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2062246499","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030526733,0.04804249,0.92053205,0.00015079696,0.0003692677,0.0001635881,0.0000026760479,0.00009890672,0.000113501905],"genre_scores_gemma":[0.5448838,0.000010006605,0.45494768,0.00006422441,0.000065623,0.0000017179249,8.72706e-7,0.000007824378,0.000018264016],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986148,0.0000483753,0.00060286035,0.0001827346,0.0003453591,0.00020590864],"domain_scores_gemma":[0.9976553,0.00008036972,0.001143714,0.00018260078,0.0008743394,0.0000637304],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005316641,0.0001570301,0.0003200879,0.00028236222,0.00008839907,0.000086537715,0.00049884967,0.00010011529,0.000003774301],"category_scores_gemma":[0.0002552972,0.000118784075,0.00014905774,0.0003597434,0.00008645147,0.0021314458,0.00007288844,0.00030246843,2.8697806e-7],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013885653,0.00004730574,0.00002940822,0.0001682275,0.000023853174,0.000030235375,0.007970262,0.0000092055825,0.011966715,0.005955367,0.000047365374,0.9736132],"study_design_scores_gemma":[0.002449863,0.0006277328,0.00017171793,0.0020455685,0.00016757975,0.0024566457,0.004624233,0.39133948,0.4457316,0.14949425,0.00008308125,0.0008082292],"about_ca_topic_score_codex":0.0000170653,"about_ca_topic_score_gemma":0.0000037359237,"teacher_disagreement_score":0.97280496,"about_ca_system_score_codex":0.000063750624,"about_ca_system_score_gemma":0.00015027974,"threshold_uncertainty_score":0.4843874},"labels":[],"label_agreement":null},{"id":"W2073971205","doi":"10.5715/jnlp.7.2_117","title":"The Exploration and Analysis of Using Multiple Thesaurus Types for Query Expansion in Information Retrieval.","year":2000,"lang":"en","type":"article","venue":"Journal of Natural Language Processing","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; York University","keywords":"Thesaurus; Information retrieval; Query expansion; Computer science; Natural language processing","score_opus":0.017030922780763687,"score_gpt":0.28357993317831415,"score_spread":0.26654901039755047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2073971205","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9353224,0.009309198,0.05494,0.00028675052,0.00006551068,0.00005716801,5.3373134e-7,0.0000064974324,0.000011943903],"genre_scores_gemma":[0.980897,0.00010721734,0.018931752,0.000035853624,0.000019839277,3.0153785e-7,9.094335e-7,0.000001296603,0.000005848379],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99939996,0.000028493068,0.0003003112,0.00004556661,0.00014737056,0.00007831249],"domain_scores_gemma":[0.9993079,0.00018457572,0.00028712605,0.00006493103,0.00014287137,0.000012599136],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004536537,0.000046907255,0.0001389502,0.00019864466,0.00007938487,0.000115601775,0.00015319472,0.00002987615,6.313993e-7],"category_scores_gemma":[0.00029020628,0.000027241675,0.000046254638,0.00047335148,0.00002048309,0.0017462733,0.000015641926,0.00008002204,6.347137e-8],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001410084,0.000009395417,0.001023126,0.000032926248,0.000035003344,0.0000024815026,0.008465014,0.002128173,0.0033483014,0.000045066234,0.000003232796,0.98476624],"study_design_scores_gemma":[0.0003244581,0.000035232486,0.008118277,0.00008900955,0.00006440552,0.000014625861,0.002093854,0.9853246,0.0035740077,0.00026030472,0.000046088593,0.000055101507],"about_ca_topic_score_codex":0.000017554878,"about_ca_topic_score_gemma":0.000028057959,"teacher_disagreement_score":0.98471117,"about_ca_system_score_codex":0.000019980156,"about_ca_system_score_gemma":0.00004321973,"threshold_uncertainty_score":0.1266006},"labels":[],"label_agreement":null},{"id":"W2316254469","doi":"10.5715/jnlp.17.2_51","title":"A Written Child Corpus with Editing History Tags","year":2010,"lang":"en","type":"article","venue":"Journal of Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Intecsea (Canada)","funders":"","keywords":"Computer science; Natural language processing; Linguistics; World Wide Web; Information retrieval; History; Philosophy","score_opus":0.004839098735990272,"score_gpt":0.23134153373138322,"score_spread":0.22650243499539294,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2316254469","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19627428,0.49943262,0.28243142,0.00798324,0.0066660126,0.00055103475,0.000003971514,0.0018476213,0.0048098],"genre_scores_gemma":[0.631804,0.0000063934144,0.36672202,0.00053831114,0.0007648815,0.0000016748343,9.269324e-7,0.000020958876,0.00014084138],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99798363,0.00004578967,0.0005056824,0.00030760374,0.0007487096,0.00040855748],"domain_scores_gemma":[0.9977173,0.00007003354,0.0010722671,0.00034041956,0.0006321551,0.00016777686],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006635141,0.00027263863,0.00037131202,0.00038448584,0.00018980494,0.00030405613,0.0015249636,0.00014725953,0.000019760548],"category_scores_gemma":[0.00036503375,0.00018285504,0.00011828551,0.0004497603,0.00012775644,0.0023640862,0.00018016018,0.002195275,0.000002686671],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008025634,0.00010323519,0.00035455488,0.00020251161,0.000043867716,0.001629265,0.0058297287,0.000002600742,0.12877995,0.0015852699,0.0015253135,0.85986346],"study_design_scores_gemma":[0.012862004,0.003445345,0.0031999361,0.012947993,0.00072777865,0.15504605,0.0048707956,0.092613764,0.5845659,0.011217026,0.110429145,0.008074293],"about_ca_topic_score_codex":0.000017792083,"about_ca_topic_score_gemma":0.00002581058,"teacher_disagreement_score":0.8517892,"about_ca_system_score_codex":0.0001715174,"about_ca_system_score_gemma":0.0003330278,"threshold_uncertainty_score":0.9537499},"labels":[],"label_agreement":null},{"id":"W3166208174","doi":"10.5715/jnlp.28.350","title":"The Effectiveness of Data Augmentation by Removing Unimportant sentence","year":2021,"lang":"en","type":"article","venue":"Journal of Natural Language Processing","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Tokyo Metropolitan University; Institute for Catastrophic Loss Reduction","keywords":"Pointer (user interface); Computer science; Generator (circuit theory); Sentence; Natural language processing; Programming language; Artificial intelligence; Physics","score_opus":0.017118238281793714,"score_gpt":0.3082740527003838,"score_spread":0.2911558144185901,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3166208174","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4151407,0.06788053,0.5158572,0.0004976267,0.00047707034,0.000053380216,0.0000024368137,0.000015673857,0.00007539454],"genre_scores_gemma":[0.95326644,0.000047385678,0.046538934,0.000048179532,0.00005703338,1.98525e-7,0.0000033260142,0.0000039631946,0.000034563138],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988203,0.0001859004,0.00034669353,0.00015379998,0.00037066857,0.00012266202],"domain_scores_gemma":[0.9985362,0.0003268273,0.00046178047,0.0003318399,0.00031085208,0.000032450105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001557391,0.00006730867,0.00013763727,0.000033514658,0.000112136455,0.00016253538,0.0008955447,0.000024163353,8.082385e-7],"category_scores_gemma":[0.0003183104,0.00004312297,0.00003466017,0.00025955474,0.00002407271,0.0012225631,0.00025925605,0.00022864601,1.5342542e-7],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048107217,0.000032932472,0.00049749186,0.0002166573,0.000035480938,0.00023808567,0.0011625409,0.00010644422,0.40672496,0.00036659604,0.000053603955,0.5905171],"study_design_scores_gemma":[0.0015397688,0.000089299734,0.0037294456,0.002576143,0.000077475386,0.0020386071,0.006483342,0.6586515,0.32271937,0.0012676326,0.00045549416,0.00037195036],"about_ca_topic_score_codex":0.000011388258,"about_ca_topic_score_gemma":0.000004924314,"teacher_disagreement_score":0.658545,"about_ca_system_score_codex":0.000053185628,"about_ca_system_score_gemma":0.0002969134,"threshold_uncertainty_score":0.17585036},"labels":[],"label_agreement":null}]}