{"meta":{"query_hash":"2026479010f0","filters":{"venue":"Natural Language Engineering"},"cohort_total":31,"direct_labels_cover":0,"predictions_cover":31,"exported":31,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/2026479010f0","api":"https://metacan.xera.ac/api/v1/cohort?venue=Natural+Language+Engineering"},"results":[{"id":"W1971541835","doi":"10.1017/s1351324901002650","title":"Real-time automatic insertion of accents in French text","year":2001,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Stress (linguistics); Character (mathematics); Word (group theory); Natural language processing; Speech recognition; Artificial intelligence; Linguistics","score_opus":0.004005314190147125,"score_gpt":0.23514128188684852,"score_spread":0.2311359676967014,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1971541835","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37409866,0.00090155145,0.50442195,0.00040042595,0.00038570666,0.00023840863,0.0009805821,0.11532991,0.0032429013],"genre_scores_gemma":[0.60856485,0.00030134374,0.3812218,0.00016510935,0.00015370714,0.0000858154,0.0014669432,0.0016079062,0.0064324653],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9982893,0.00038735272,0.00013042314,0.00062742137,0.00042655758,0.00013899038],"domain_scores_gemma":[0.99379265,0.0030624892,0.0007950116,0.0008634774,0.0012707309,0.0002156601],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014012033,0.0011549917,0.0009666161,0.0011740486,0.0006261873,0.0014268197,0.00119014,0.0010311931,0.0030241064],"category_scores_gemma":[0.0068376646,0.000530306,0.0004558305,0.00075581315,0.00049712486,0.001312222,0.0007186154,0.000709944,0.003942902],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014181717,0.00016912546,0.004771527,0.0003489088,0.000051880153,0.00067572587,0.0012338068,0.0048425416,0.24648944,0.0006641109,0.0076667136,0.73166806],"study_design_scores_gemma":[0.0001390378,0.0011224849,0.017320763,0.00005542443,0.00015419861,0.0029356822,0.00071048085,0.4064515,0.53901285,0.0015735055,0.030242985,0.00028109318],"about_ca_topic_score_codex":0.0033478497,"about_ca_topic_score_gemma":0.0031293465,"teacher_disagreement_score":0.0033478497,"about_ca_system_score_codex":0.00041222555,"about_ca_system_score_gemma":0.00039457693,"threshold_uncertainty_score":0.010116637},"labels":[],"label_agreement":null},{"id":"W2029070039","doi":"10.1017/s135132490600444x","title":"A general feature space for automatic verb classification","year":2006,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Lexicon; Artificial intelligence; Feature vector; Natural language processing; Feature (linguistics); Redundancy (engineering); Verb; Support vector machine; Variety (cybernetics); Feature selection; Machine learning; Linguistics","score_opus":0.004724068616820104,"score_gpt":0.235974427016896,"score_spread":0.2312503584000759,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2029070039","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044369638,0.00027851967,0.9502981,0.00016974821,0.000037572052,0.00010488867,0.000541235,0.0036016258,0.00059860706],"genre_scores_gemma":[0.5600121,0.00014980524,0.43583864,0.00010141822,0.00005618458,0.00043437196,0.0020776591,0.00017356992,0.00115623],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987551,0.0004867452,0.000114606795,0.00024283855,0.00029761318,0.00010305925],"domain_scores_gemma":[0.9972331,0.0015768723,0.00016602242,0.00034270325,0.000616604,0.00006459888],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018648098,0.0009028724,0.0010007242,0.0018399539,0.00045359973,0.0012136168,0.0011235698,0.0009818068,0.0030435824],"category_scores_gemma":[0.00506897,0.00030235937,0.00095736916,0.001673197,0.00062947563,0.0019636622,0.0011074941,0.0010151356,0.0010098986],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055014394,0.00030495858,0.002765108,0.00020595189,0.00008313221,0.00012717948,0.00009866509,0.10855377,0.020340905,0.009618376,0.007277318,0.8500745],"study_design_scores_gemma":[0.0000258956,0.00010972236,0.0008984255,0.000019653195,0.000013418455,0.00006353913,0.000029687442,0.98262066,0.005804028,0.008624844,0.0017711753,0.000019038669],"about_ca_topic_score_codex":0.0021091036,"about_ca_topic_score_gemma":0.00124882,"teacher_disagreement_score":0.0030435824,"about_ca_system_score_codex":0.00064588286,"about_ca_system_score_gemma":0.0007808309,"threshold_uncertainty_score":0.010181844},"labels":[],"label_agreement":null},{"id":"W2047851837","doi":"10.1017/s1351324901002716","title":"Scalable generation of texts using causal and temporal expansions of sentences","year":2001,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Sentence; Artificial intelligence; Natural language processing; Scalability; Kernel (algebra); Process (computing); Theoretical computer science; Information retrieval; Programming language","score_opus":0.015037052180989056,"score_gpt":0.2628490486382299,"score_spread":0.24781199645724084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2047851837","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028933601,0.00013701833,0.9600132,0.00018481242,0.000041924028,0.00023923807,0.00026349412,0.007584094,0.0026025737],"genre_scores_gemma":[0.19716692,0.00014056252,0.7973164,0.00008856987,0.00007117372,0.00024081435,0.0011795848,0.0008036014,0.002992435],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986628,0.00048575984,0.000084530504,0.0003033127,0.0004088757,0.000054754197],"domain_scores_gemma":[0.9941606,0.0041515124,0.0002458232,0.0008522181,0.00050336425,0.00008648098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017411901,0.0006828542,0.00052166014,0.00089965854,0.0005392498,0.0009581512,0.0012793147,0.00053715194,0.00507691],"category_scores_gemma":[0.009235212,0.00045821044,0.0008970586,0.0005578607,0.0007526568,0.0024847954,0.0016697994,0.00087999477,0.0012886979],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007519955,0.00030475823,0.0017102312,0.0006967605,0.00011228829,0.0009818812,0.0019867902,0.06566448,0.10696226,0.052806538,0.009141457,0.75888056],"study_design_scores_gemma":[0.00017894166,0.00026997412,0.0013482964,0.000073764386,0.00017507224,0.0005812069,0.00039662662,0.8070766,0.09435052,0.06295892,0.03250411,0.00008593037],"about_ca_topic_score_codex":0.00076267467,"about_ca_topic_score_gemma":0.0012331394,"teacher_disagreement_score":0.00507691,"about_ca_system_score_codex":0.0004043153,"about_ca_system_score_gemma":0.00047381842,"threshold_uncertainty_score":0.016983986},"labels":[],"label_agreement":null},{"id":"W2052696540","doi":"10.1017/s1351324908005044","title":"A corpus-based analysis of argument realization by preposition structures","year":2009,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"FrameNet; Computer science; Realization (probability); Argument (complex analysis); Natural language processing; Artificial intelligence; Semantics (computer science); Semantic role labeling; Linguistics; Frame (networking); Parsing; Programming language; Sentence; Mathematics; Philosophy","score_opus":0.0024766510758832297,"score_gpt":0.231932056132034,"score_spread":0.22945540505615075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2052696540","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8877897,0.0017132261,0.0697191,0.00050305197,0.00019702935,0.00027065453,0.017517302,0.0006122798,0.021677623],"genre_scores_gemma":[0.90894353,0.0008246829,0.06358326,0.000055181226,0.000052255196,0.00036083843,0.023630204,0.00024011979,0.002309964],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984251,0.000729441,0.00013763069,0.00028667058,0.0003763405,0.000044818284],"domain_scores_gemma":[0.98306227,0.012691675,0.00079582515,0.0018417927,0.001497097,0.00011136005],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020926541,0.0002466572,0.00037110352,0.0053412947,0.0014625423,0.001467859,0.0004965686,0.0005144405,0.004040629],"category_scores_gemma":[0.012617771,0.00036250486,0.00031033225,0.007999713,0.0011669007,0.0018978646,0.0011262504,0.0009275149,0.00076431973],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002974237,0.0010458942,0.08902512,0.00493756,0.00053556677,0.004548921,0.037271217,0.01724724,0.11396074,0.15616016,0.045812294,0.52648103],"study_design_scores_gemma":[0.00031679028,0.00055795006,0.37615746,0.0010032183,0.00052121986,0.0050923782,0.022471808,0.09639845,0.098318085,0.046141587,0.35265526,0.00036582153],"about_ca_topic_score_codex":0.003706906,"about_ca_topic_score_gemma":0.0068263696,"teacher_disagreement_score":0.0053412947,"about_ca_system_score_codex":0.0008323531,"about_ca_system_score_gemma":0.0007376922,"threshold_uncertainty_score":0.013517261},"labels":[],"label_agreement":null},{"id":"W2058214962","doi":"10.1017/s1351324913000090","title":"On the semantics of noun compounds","year":2013,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Automatic summarization; Computer science; Noun; Natural language processing; Cover (algebra); Artificial intelligence; Proper noun; Semantics (computer science); Subject (documents); Machine translation; Linguistics; Noun phrase; Question answering; Sequence (biology); World Wide Web; Programming language; Philosophy; Chemistry","score_opus":0.0038637183364422247,"score_gpt":0.2078067341440626,"score_spread":0.20394301580762036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2058214962","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035982788,0.032373566,0.72077465,0.02252554,0.0024811432,0.00018013013,0.0013525173,0.0010387149,0.18329091],"genre_scores_gemma":[0.6532979,0.023248617,0.2758824,0.006891634,0.004221934,0.00050239795,0.002474962,0.0015295694,0.031950567],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9972754,0.0011368357,0.000332074,0.00054712506,0.00053760293,0.00017103343],"domain_scores_gemma":[0.99465454,0.003259754,0.00031780923,0.00064821675,0.0009365739,0.00018308908],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031815502,0.0010644068,0.0010612903,0.004041028,0.004044357,0.007120886,0.0018640512,0.0024824883,0.0072871526],"category_scores_gemma":[0.008473878,0.0010529693,0.0014643687,0.0042169876,0.01209594,0.028824419,0.0042313486,0.0043413863,0.0022438061],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002055672,0.000009426374,0.00010721717,0.00006747326,0.000008885433,0.000112901,0.00091268896,0.00057538773,0.00040525864,0.98638505,0.0026913686,0.008703816],"study_design_scores_gemma":[0.0000074059158,0.000008140606,0.00009701839,0.00006213915,0.000008317053,0.00011100302,0.00027357554,0.0019086509,0.00029058105,0.9665482,0.03067159,0.000013444649],"about_ca_topic_score_codex":0.0045205005,"about_ca_topic_score_gemma":0.003227541,"teacher_disagreement_score":0.0072871526,"about_ca_system_score_codex":0.0026225497,"about_ca_system_score_gemma":0.0015644699,"threshold_uncertainty_score":0.024377942},"labels":[],"label_agreement":null},{"id":"W2089752704","doi":"10.1017/s1351324910000264","title":"Subjectivity detection in spoken and written conversations","year":2010,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Conversation; Set (abstract data type); Subjectivity; Natural language processing; Artificial intelligence; Domain (mathematical analysis); Speech recognition; Linguistics","score_opus":0.0024433274761154348,"score_gpt":0.20363026364576112,"score_spread":0.20118693616964567,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2089752704","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9308406,0.00045556587,0.0595058,0.00022151662,0.000110615336,0.00018729971,0.00085004803,0.0009259195,0.006902582],"genre_scores_gemma":[0.979815,0.00014042175,0.017580839,0.000054314834,0.00008910652,0.00007944891,0.0007097833,0.00006454933,0.0014664215],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977198,0.001128231,0.00014447402,0.00041467632,0.00040153018,0.00019117788],"domain_scores_gemma":[0.98717016,0.008338367,0.0015094599,0.0006451843,0.0018379168,0.0004989072],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018830758,0.0005442384,0.00061248423,0.0019630857,0.00048005368,0.0016746187,0.0004079836,0.00062627206,0.0017523441],"category_scores_gemma":[0.012242606,0.00023860768,0.00042855943,0.0006342608,0.00038367874,0.0017208076,0.0013602284,0.000666285,0.0010358167],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003731202,0.00046732792,0.1683859,0.0010067945,0.00035210894,0.0009166583,0.006668733,0.0068836175,0.21739371,0.0015813542,0.0043892185,0.58822334],"study_design_scores_gemma":[0.00013394309,0.001170542,0.42909873,0.00026863898,0.00030153323,0.0016408195,0.008980107,0.37087902,0.16555476,0.00974896,0.011926938,0.00029603855],"about_ca_topic_score_codex":0.0010300669,"about_ca_topic_score_gemma":0.0012600559,"teacher_disagreement_score":0.0019630857,"about_ca_system_score_codex":0.0003176414,"about_ca_system_score_gemma":0.0003263499,"threshold_uncertainty_score":0.009958744},"labels":[],"label_agreement":null},{"id":"W2101481293","doi":"10.1017/s1351324905003694","title":"Segmenting documents by stylistic character","year":2005,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":104,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Character (mathematics); Bigram; Natural language processing; Artificial intelligence; Vocabulary; Baseline (sea); Artificial neural network; Market segmentation; Information retrieval; Linguistics; Trigram","score_opus":0.002330387928883056,"score_gpt":0.22306770749537055,"score_spread":0.2207373195664875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2101481293","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.80492085,0.0014039766,0.17436254,0.000438691,0.00022833256,0.0006644359,0.004186722,0.005508958,0.008285484],"genre_scores_gemma":[0.75067335,0.00046296327,0.2357334,0.00006779058,0.00007567023,0.00012488423,0.006895785,0.00030616412,0.005659995],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99942136,0.00011499055,0.000074055235,0.00022136407,0.00010091105,0.00006726697],"domain_scores_gemma":[0.996298,0.0017672394,0.00041560738,0.000437721,0.00094255776,0.000138972],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00065757526,0.0007604807,0.00043019973,0.0025816236,0.00059237814,0.0014711111,0.0004751205,0.00069822604,0.0025702042],"category_scores_gemma":[0.0049259616,0.00024102426,0.00033544863,0.0018962693,0.00026854122,0.0015613508,0.00040190015,0.0007431085,0.0018414679],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014663542,0.00024349644,0.019756313,0.0005982485,0.00007541082,0.00054131116,0.0015391299,0.01635694,0.15219797,0.002612825,0.007200539,0.7974115],"study_design_scores_gemma":[0.00015091925,0.0005785497,0.03348251,0.00013420911,0.00015461729,0.0011797794,0.0026011274,0.6907223,0.2141464,0.013176785,0.043532558,0.00014032517],"about_ca_topic_score_codex":0.0047638323,"about_ca_topic_score_gemma":0.008074566,"teacher_disagreement_score":0.0047638323,"about_ca_system_score_codex":0.000708891,"about_ca_system_score_gemma":0.000600246,"threshold_uncertainty_score":0.009472251},"labels":[],"label_agreement":null},{"id":"W2121929533","doi":"10.1017/s135132491300003x","title":"Designing a machine translation system for Canadian weather warnings: A case study","year":2013,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal; Université du Québec à Montréal","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Metric (unit); Task (project management); Quality (philosophy); Domain (mathematical analysis); Evaluation of machine translation; Translation (biology); Artificial intelligence; Machine translation software usability; Presentation (obstetrics); Example-based machine translation; Machine learning; Information retrieval; Systems engineering","score_opus":0.007204560510228801,"score_gpt":0.2309438070816249,"score_spread":0.2237392465713961,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121929533","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.73459977,0.001326954,0.17345683,0.006377378,0.0007575516,0.003353867,0.0066814437,0.01415146,0.059294738],"genre_scores_gemma":[0.7016811,0.0008062366,0.24925809,0.00061286945,0.00009801871,0.00041512426,0.00822524,0.0014401025,0.03746321],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9975339,0.0008231773,0.00016741462,0.00032483868,0.00088698504,0.00026367797],"domain_scores_gemma":[0.9959973,0.0013383925,0.00013370282,0.00029101182,0.0019692103,0.00027037834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027109513,0.0009796236,0.00055642024,0.0011394726,0.004927296,0.0019204371,0.0011298265,0.0015872229,0.0059156884],"category_scores_gemma":[0.007593164,0.00034825347,0.00053884904,0.002144947,0.0012425245,0.0013677087,0.00081979576,0.0013516163,0.002190729],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012587791,0.0015573917,0.018690666,0.0035643186,0.00016400634,0.017066132,0.03291257,0.064693995,0.15131333,0.01636749,0.0978449,0.5945664],"study_design_scores_gemma":[0.0009629357,0.0012806387,0.042835575,0.00035451984,0.00036597726,0.0056727096,0.024682162,0.1524724,0.23042305,0.0034870063,0.536755,0.00070803793],"about_ca_topic_score_codex":0.521345,"about_ca_topic_score_gemma":0.54722166,"teacher_disagreement_score":0.47865498,"about_ca_system_score_codex":0.008337611,"about_ca_system_score_gemma":0.014768796,"threshold_uncertainty_score":0.9629477},"labels":[],"label_agreement":null},{"id":"W2126034021","doi":"10.1017/s1351324901002765","title":"Discovery of inference rules for question-answering","year":2001,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":532,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Question answering; Inference; Computer science; Dependency (UML); Set (abstract data type); Parsing; Artificial intelligence; Natural language processing; Rule of inference; Programming language","score_opus":0.00715054306742324,"score_gpt":0.2517287714730841,"score_spread":0.24457822840566085,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2126034021","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0067988033,0.0003653209,0.9827832,0.00080081896,0.00005199747,0.0003577637,0.0009942934,0.006214378,0.0016333653],"genre_scores_gemma":[0.05577256,0.0003113596,0.9375631,0.00033869751,0.000121391924,0.00037481636,0.004171626,0.00039183703,0.0009544628],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9877671,0.004144249,0.0014878274,0.0032813326,0.002952466,0.00036705678],"domain_scores_gemma":[0.9239963,0.060990427,0.0026269376,0.006629165,0.005140124,0.0006169625],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01285425,0.0016765106,0.001928733,0.007963133,0.0021711406,0.003862006,0.004943471,0.0028762098,0.004924751],"category_scores_gemma":[0.08408172,0.0016174404,0.0034202896,0.003867732,0.002223716,0.009593174,0.0038252957,0.0049922033,0.0038363792],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029533787,0.0006702197,0.014070659,0.0012709078,0.00049688324,0.001019094,0.0034075996,0.026488563,0.011181616,0.11923913,0.028005902,0.7938541],"study_design_scores_gemma":[0.00008116165,0.00005574021,0.0024393387,0.00024227216,0.00022057795,0.0006252131,0.0005403052,0.5626093,0.014832978,0.3893543,0.02891343,0.000085374355],"about_ca_topic_score_codex":0.0038652981,"about_ca_topic_score_gemma":0.0053856303,"teacher_disagreement_score":0.01285425,"about_ca_system_score_codex":0.0016318441,"about_ca_system_score_gemma":0.0026073796,"threshold_uncertainty_score":0.06798059},"labels":[],"label_agreement":null},{"id":"W2128424286","doi":"10.1017/s1351324911000167","title":"Query-focused multi-document summarization: automatic data annotations and supervised learning approaches","year":2011,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Automatic summarization; Artificial intelligence; Conditional random field; Semi-supervised learning; Supervised learning; Support vector machine; Annotation; Machine learning; Information retrieval; Natural language processing; Artificial neural network","score_opus":0.05869544911709373,"score_gpt":0.23725411456844478,"score_spread":0.17855866545135105,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128424286","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.045954797,0.0012153274,0.9453755,0.0003531661,0.00006171583,0.00020563371,0.00031595657,0.0057233577,0.00079456466],"genre_scores_gemma":[0.28471115,0.00042646716,0.7108411,0.00015304884,0.00018756583,0.00026469698,0.001936125,0.00030608964,0.0011738337],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9946114,0.0030173871,0.00037861546,0.0009843139,0.0008850521,0.00012321462],"domain_scores_gemma":[0.9788084,0.011889986,0.0022405484,0.0027872499,0.003968779,0.00030505154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053687473,0.001290555,0.0013351663,0.0033314594,0.0008739584,0.0012193682,0.0016706086,0.0011543587,0.0007106849],"category_scores_gemma":[0.015067327,0.0004132582,0.0009309745,0.002231134,0.00059046806,0.002876069,0.0010589629,0.0013578049,0.0007266281],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005805876,0.00073128566,0.0038840824,0.0006151026,0.00032027016,0.00012340664,0.00062352355,0.07474454,0.02671475,0.0021542676,0.004666821,0.8848415],"study_design_scores_gemma":[0.00006551621,0.00034503042,0.0030827601,0.000056201527,0.00016547575,0.00011343441,0.00020584618,0.9503317,0.035996318,0.0060813823,0.0034919474,0.00006445294],"about_ca_topic_score_codex":0.0021132242,"about_ca_topic_score_gemma":0.0043288297,"teacher_disagreement_score":0.0053687473,"about_ca_system_score_codex":0.0007236545,"about_ca_system_score_gemma":0.00090369175,"threshold_uncertainty_score":0.02839303},"labels":[],"label_agreement":null},{"id":"W2129174913","doi":"10.1017/s1351324908004683","title":"Industry Watch: Language technology, meet social networking","year":2008,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Service-Oriented Architecture and Web Services","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Thread (computing); World Wide Web; Quarter (Canadian coin); Language technology; Data science; Telecommunications; Artificial intelligence; Natural language; Programming language; Comprehension approach; History","score_opus":0.0046112725476018706,"score_gpt":0.20963480101097304,"score_spread":0.20502352846337116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129174913","genre_codex":"other","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01157026,0.0113241365,0.0042321403,0.14395879,0.030551683,0.00021842106,0.005436687,0.009862799,0.78284496],"genre_scores_gemma":[0.0356392,0.0053691147,0.0028552175,0.028215785,0.012681012,0.00017767173,0.0045638476,0.0017724889,0.9087256],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996872,0.000044156386,0.00001144069,0.000033499247,0.0001359604,0.000087797285],"domain_scores_gemma":[0.9988727,0.00015529999,0.00006968733,0.000054045802,0.00018639198,0.00066198065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007179299,0.00075001514,0.00025961327,0.0015231277,0.0024997934,0.003911814,0.0005639233,0.0022494562,0.17328387],"category_scores_gemma":[0.0023914601,0.0002420449,0.00033588204,0.000988118,0.00039325794,0.0055354564,0.002156726,0.0021939853,0.09054659],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016866714,0.000018937513,0.00025542764,0.000037657293,0.0000022943664,0.00004258333,0.00014156889,0.0000073641386,0.00031238812,0.001178311,0.97272724,0.025259307],"study_design_scores_gemma":[0.0000076269744,0.00002308988,0.0009680658,0.00003170166,0.0000038768208,0.000058641926,0.00064753403,0.000072627925,0.00017711554,0.0006534039,0.9973484,0.000007948607],"about_ca_topic_score_codex":0.0023308585,"about_ca_topic_score_gemma":0.008295477,"teacher_disagreement_score":0.17328387,"about_ca_system_score_codex":0.000526891,"about_ca_system_score_gemma":0.00052079966,"threshold_uncertainty_score":0.5796923},"labels":[],"label_agreement":null},{"id":"W2129913068","doi":"10.1017/s135132491100012x","title":"Learning opinions in user-generated web content","year":2011,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Children's Hospital of Eastern Ontario; University of Ottawa","funders":"","keywords":"Computer science; Construct (python library); Information retrieval; Product (mathematics); Hierarchy; User-generated content; Natural language processing; Sentiment analysis; World Wide Web; Artificial intelligence; Web content; Web page; Social media","score_opus":0.022416421518357734,"score_gpt":0.22655476296845853,"score_spread":0.2041383414501008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129913068","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9655349,0.00014482455,0.0319426,0.00018924185,0.000034745364,0.000086699336,0.00047319283,0.00024300431,0.0013507277],"genre_scores_gemma":[0.99008596,0.000039810282,0.00870347,0.000026529886,0.00003648583,0.000026394267,0.0007265876,0.000012356099,0.00034246076],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985373,0.0007574063,0.00008613331,0.0002296275,0.00029721533,0.000092300186],"domain_scores_gemma":[0.98777664,0.00842526,0.0011260076,0.0003295978,0.0021154424,0.00022701758],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018496959,0.00043109283,0.00035654465,0.0017932327,0.00018758161,0.0011051855,0.00038543917,0.00060212973,0.00077094085],"category_scores_gemma":[0.015158251,0.00015734682,0.0003599691,0.00080290943,0.00024186035,0.0012702123,0.00031139774,0.0005389194,0.00047305683],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023651067,0.001708509,0.3310988,0.0004893154,0.0006008016,0.0009332584,0.0020124726,0.0947165,0.034988213,0.0024367692,0.007866274,0.520784],"study_design_scores_gemma":[0.000020812404,0.00023778203,0.047842212,0.000018829842,0.000048299094,0.000072255025,0.00030040587,0.9432367,0.0059626694,0.0015590069,0.0006785405,0.000022422726],"about_ca_topic_score_codex":0.0016891935,"about_ca_topic_score_gemma":0.001800506,"teacher_disagreement_score":0.0018496959,"about_ca_system_score_codex":0.0006769084,"about_ca_system_score_gemma":0.00018902891,"threshold_uncertainty_score":0.009782255},"labels":[],"label_agreement":null},{"id":"W2136433925","doi":"10.1017/s1351324903003231","title":"Surface-marker-based dialog modelling: A progress report on the MAREDI project","year":2003,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; Université Laval; Université du Québec à Trois-Rivières","funders":"","keywords":"Computer science; Dialog box; Natural language processing; Connectionism; Semantics (computer science); Conversation; Spoken language; Artificial intelligence; Natural language; Natural language understanding; Natural language generation; Programming language; Linguistics; Artificial neural network; World Wide Web","score_opus":0.01414064130185601,"score_gpt":0.25566720439021334,"score_spread":0.24152656308835732,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136433925","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03500116,0.0035100328,0.9349026,0.0010096027,0.00017948388,0.0006575201,0.001243248,0.014041256,0.009455197],"genre_scores_gemma":[0.13023801,0.0026641057,0.85166967,0.00018959084,0.0001872902,0.00043092112,0.0034290235,0.0014295374,0.0097617535],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970733,0.0011696123,0.00015383204,0.0007974995,0.00068869075,0.000117154086],"domain_scores_gemma":[0.9958973,0.0016654773,0.00019415256,0.001286248,0.00072965934,0.00022729753],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052035097,0.0015665464,0.0014762776,0.0017120636,0.00060506247,0.0036957436,0.0039170315,0.0017609124,0.006767148],"category_scores_gemma":[0.009961942,0.0011777737,0.0013175085,0.0008961253,0.0012373362,0.0069303336,0.0020828112,0.0022872994,0.0030205073],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090643734,0.0009860353,0.003115988,0.001627226,0.00022613545,0.00038540852,0.002563362,0.108362615,0.055373468,0.08220652,0.017300418,0.7269464],"study_design_scores_gemma":[0.00014389287,0.0003798252,0.0023206018,0.00021521507,0.00014015628,0.0004373496,0.0003914485,0.8023509,0.04921948,0.018922288,0.12527758,0.00020135191],"about_ca_topic_score_codex":0.008152458,"about_ca_topic_score_gemma":0.0032547617,"teacher_disagreement_score":0.008152458,"about_ca_system_score_codex":0.0014875556,"about_ca_system_score_gemma":0.002075558,"threshold_uncertainty_score":0.027519166},"labels":[],"label_agreement":null},{"id":"W2137926008","doi":"10.1017/s1351324905004043","title":"Can syllabification improve pronunciation by analogy of English?","year":2006,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; National Research Council Institute for Biodiagnostics","funders":"","keywords":"Syllabification; Pronunciation; Computer science; Syllable; Natural language processing; Analogy; Artificial intelligence; Speech recognition; Word (group theory); Linguistics","score_opus":0.0016253391825071673,"score_gpt":0.19236527812346438,"score_spread":0.1907399389409572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2137926008","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3975843,0.0032571263,0.5695746,0.0025987276,0.00069905655,0.00014657396,0.00033631135,0.00718846,0.018614918],"genre_scores_gemma":[0.8140182,0.0005376015,0.18198249,0.000415823,0.000103775674,0.000049748818,0.00036084594,0.00024891165,0.0022826544],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99850684,0.00068921404,0.00007182569,0.00046341968,0.00017553741,0.00009319487],"domain_scores_gemma":[0.9940608,0.00391124,0.00027649713,0.0010301949,0.0005707787,0.00015044548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002164605,0.0007957246,0.0008982863,0.00085210905,0.00061301305,0.0015083201,0.0012912552,0.001068291,0.0037043004],"category_scores_gemma":[0.018504255,0.00040683014,0.0008585754,0.00074695697,0.00056700374,0.0051240255,0.0014417521,0.0014874849,0.0020234354],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007694302,0.00028388857,0.019534815,0.00029020727,0.00013305894,0.0002461776,0.0009810155,0.021254595,0.032567434,0.016729238,0.002895321,0.9043148],"study_design_scores_gemma":[0.00015623667,0.0010513826,0.023171922,0.0001496012,0.00028511524,0.0011123723,0.000998642,0.7651165,0.062378626,0.1202098,0.02514736,0.00022246914],"about_ca_topic_score_codex":0.0034749762,"about_ca_topic_score_gemma":0.0040370957,"teacher_disagreement_score":0.0037043004,"about_ca_system_score_codex":0.00060613925,"about_ca_system_score_gemma":0.0010311778,"threshold_uncertainty_score":0.012392163},"labels":[],"label_agreement":null},{"id":"W2148362501","doi":"10.1017/s1351324904003560","title":"Correcting real-word spelling errors by restoring lexical cohesion","year":2005,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":180,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Spelling; Computer science; Cohesion (chemistry); Natural language processing; Lexicon; Artificial intelligence; Word (group theory); Context (archaeology); Precision and recall; Recall; Speech recognition; Linguistics","score_opus":0.005958882391220053,"score_gpt":0.2512037126170917,"score_spread":0.24524483022587162,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2148362501","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.52111965,0.0013645913,0.4521361,0.0006001048,0.0004623499,0.0003287003,0.0009890244,0.018496858,0.0045025805],"genre_scores_gemma":[0.611146,0.0004590046,0.38164377,0.00018862175,0.00011550092,0.00011125076,0.0015594782,0.0011624002,0.0036139956],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99766064,0.0003889921,0.00035226595,0.0006811218,0.00079342007,0.0001235423],"domain_scores_gemma":[0.9816126,0.0042293463,0.0035818592,0.0062554227,0.004041274,0.0002794037],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018421201,0.0010673602,0.0013143965,0.003202153,0.0008274391,0.0014600368,0.0011778628,0.0009905815,0.001872053],"category_scores_gemma":[0.020042837,0.00046553067,0.00058420375,0.0021450154,0.00083409343,0.002031855,0.0016834474,0.00093600544,0.0020656746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033972747,0.00030967648,0.01995013,0.0007794987,0.00015451158,0.0008543896,0.0013805039,0.008452491,0.1991465,0.0021699988,0.005659628,0.76080304],"study_design_scores_gemma":[0.00031438094,0.00095836207,0.06800993,0.00026335762,0.0007602337,0.0054619336,0.0016685543,0.16688253,0.6866637,0.018787794,0.049793266,0.0004359946],"about_ca_topic_score_codex":0.0022954163,"about_ca_topic_score_gemma":0.0038770663,"teacher_disagreement_score":0.003202153,"about_ca_system_score_codex":0.00040201415,"about_ca_system_score_gemma":0.0013486904,"threshold_uncertainty_score":0.0097422},"labels":[],"label_agreement":null},{"id":"W2157616430","doi":"10.1017/s1351324908004737","title":"Multilingual pronunciation by analogy","year":2008,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Institute for Biodiagnostics","funders":"","keywords":"Computer science; Pronunciation; Transcription (linguistics); Natural language processing; Orthography; Spelling; Analogy; Artificial intelligence; Variation (astronomy); Phonetic transcription; Linguistics; Speech recognition","score_opus":0.004698680563171272,"score_gpt":0.2267604826131823,"score_spread":0.22206180205001103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157616430","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6555782,0.0014106855,0.3097917,0.00025867068,0.00022130254,0.00019827428,0.0019414026,0.005632005,0.024967747],"genre_scores_gemma":[0.93966454,0.00016413188,0.055126455,0.00004866818,0.000029924548,0.000050879684,0.0018230415,0.00024709635,0.0028451795],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99728644,0.0011520714,0.00022424416,0.0007270664,0.00046740417,0.00014272447],"domain_scores_gemma":[0.9961959,0.0019059984,0.00019533848,0.00076665846,0.0008723659,0.000063718224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011998158,0.0006079079,0.0006839917,0.001232567,0.0004237048,0.0010534027,0.00050859747,0.00029589448,0.0069946847],"category_scores_gemma":[0.0062983003,0.0001882266,0.0003667937,0.0012029891,0.00037004918,0.0010190159,0.0017906637,0.00051546854,0.0031241502],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090464606,0.00016098455,0.024277382,0.00065297866,0.00015599036,0.0008648525,0.0016658352,0.030837726,0.10652626,0.0039734077,0.0034569332,0.82652295],"study_design_scores_gemma":[0.00034077402,0.00244945,0.11389663,0.00027897907,0.00033396145,0.007825371,0.004455219,0.42349657,0.31346577,0.024346707,0.10881224,0.00029844747],"about_ca_topic_score_codex":0.0024631429,"about_ca_topic_score_gemma":0.0027775576,"teacher_disagreement_score":0.0069946847,"about_ca_system_score_codex":0.00034269423,"about_ca_system_score_gemma":0.00043474344,"threshold_uncertainty_score":0.023399591},"labels":[],"label_agreement":null},{"id":"W2159087629","doi":"10.1017/s1351324911000118","title":"A hierarchical approach to mood classification in blogs","year":2011,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Task (project management); Mood; Hierarchy; Set (abstract data type); Artificial intelligence; Machine learning; Sentiment analysis; Training set; Orientation (vector space); Natural language processing; Data set; Psychology","score_opus":0.01986886967877055,"score_gpt":0.23441606871738657,"score_spread":0.21454719903861602,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159087629","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17277862,0.0016608939,0.80581707,0.0010359172,0.0003098799,0.00063697237,0.0032447756,0.0045899246,0.009926039],"genre_scores_gemma":[0.65030545,0.00032882264,0.34102142,0.0001837088,0.00029991227,0.0003539719,0.0035221737,0.00018991182,0.003794614],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985702,0.00051949866,0.00012596771,0.0003350371,0.0002834774,0.00016587762],"domain_scores_gemma":[0.99698323,0.0012684517,0.0002910272,0.00034951683,0.0009519132,0.0001558324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018033439,0.0006212797,0.00070317695,0.004992743,0.0013046992,0.0014658804,0.00097228686,0.00065964606,0.0033661597],"category_scores_gemma":[0.0057584825,0.0003640403,0.000856318,0.002895956,0.00047230555,0.0014008598,0.001161119,0.001118101,0.0016420824],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005427099,0.0007141877,0.028541606,0.00039317805,0.00018390208,0.00022826994,0.0007308711,0.021800796,0.026189126,0.0070070117,0.024741286,0.8889271],"study_design_scores_gemma":[0.00007053383,0.00015315299,0.023979135,0.00007315566,0.00009569492,0.00012141468,0.0003889947,0.93334395,0.007116538,0.02814004,0.006457412,0.00005996676],"about_ca_topic_score_codex":0.0075345226,"about_ca_topic_score_gemma":0.01445476,"teacher_disagreement_score":0.0075345226,"about_ca_system_score_codex":0.00086483563,"about_ca_system_score_gemma":0.0008609396,"threshold_uncertainty_score":0.014981329},"labels":[],"label_agreement":null},{"id":"W2160938081","doi":"10.1017/s1351324912000289","title":"Modeling human newspaper readers: The Fuzzy Believer approach","year":2012,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Newspaper; Computer science; Fuzzy logic; Set (abstract data type); Artificial intelligence; Range (aeronautics); Natural (archaeology); Fuzzy set; Information extraction; Natural language processing; Information retrieval; Data mining; Advertising; Programming language","score_opus":0.015391496316516116,"score_gpt":0.2320307436379919,"score_spread":0.21663924732147577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2160938081","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3882325,0.00036803848,0.60203904,0.0013649905,0.000027653923,0.00019641133,0.00037694164,0.0008424654,0.006551895],"genre_scores_gemma":[0.9150895,0.00009486881,0.08322735,0.000079086,0.000035959616,0.000056111072,0.00016512044,0.00002129115,0.0012307334],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99835736,0.0008973602,0.0000639302,0.0003674797,0.00024497163,0.00006896384],"domain_scores_gemma":[0.9914575,0.00686009,0.00057126075,0.00040791568,0.0005336907,0.00016969092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031789837,0.0004943593,0.0004950145,0.0017747207,0.00047645663,0.0019213824,0.001207176,0.0011863862,0.0025026987],"category_scores_gemma":[0.011057295,0.00037663965,0.0006846834,0.00059246994,0.0006583535,0.0018995977,0.00066801475,0.00070998294,0.00060688436],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014422686,0.0012408026,0.111652516,0.00046734646,0.00094583625,0.0016116904,0.012025465,0.36153916,0.025820969,0.06562392,0.004946917,0.4126831],"study_design_scores_gemma":[0.000022453916,0.00012287813,0.004173066,0.00001691926,0.000069004025,0.00013032246,0.0005411913,0.9762547,0.0016532294,0.01610307,0.000884874,0.00002823764],"about_ca_topic_score_codex":0.0034821038,"about_ca_topic_score_gemma":0.0030747293,"teacher_disagreement_score":0.0034821038,"about_ca_system_score_codex":0.00064874935,"about_ca_system_score_gemma":0.00034873825,"threshold_uncertainty_score":0.016812265},"labels":[],"label_agreement":null},{"id":"W2167742478","doi":"10.1017/s1351324911000210","title":"Exploring patterns in dictionary definitions for synonym extraction","year":2011,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Synonym (taxonomy); Natural language processing; Artificial intelligence; Lexicon; Quality (philosophy)","score_opus":0.07898671688241453,"score_gpt":0.262412322812195,"score_spread":0.18342560592978047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167742478","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17149584,0.0022649944,0.80573595,0.000957816,0.0002046421,0.000964707,0.00617021,0.0047111576,0.0074947183],"genre_scores_gemma":[0.33989075,0.0007448734,0.651971,0.00011062637,0.00005296311,0.0002647203,0.005856051,0.0002179523,0.00089106837],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99657464,0.0012657958,0.00065408606,0.00076917006,0.0006452357,0.00009114416],"domain_scores_gemma":[0.991148,0.005068125,0.0011694964,0.0010528294,0.0013550505,0.00020648392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014325291,0.0007350294,0.0008512812,0.006610184,0.00065042306,0.0019610436,0.0012294223,0.00075102755,0.0037047141],"category_scores_gemma":[0.012591949,0.00036080924,0.0006039606,0.0060331165,0.00062089495,0.0037805948,0.0015385125,0.0008603398,0.0020292404],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035522858,0.00038500156,0.020545712,0.002177106,0.00023360724,0.00096183456,0.0013020609,0.0036922437,0.05378551,0.011541228,0.010354377,0.89466614],"study_design_scores_gemma":[0.0004227407,0.0011157221,0.054256026,0.001331186,0.0006685416,0.010483638,0.007544276,0.49331415,0.18670344,0.08120048,0.16261695,0.00034281076],"about_ca_topic_score_codex":0.00065900356,"about_ca_topic_score_gemma":0.0015901482,"teacher_disagreement_score":0.006610184,"about_ca_system_score_codex":0.00037710817,"about_ca_system_score_gemma":0.0010452305,"threshold_uncertainty_score":0.012393534},"labels":[],"label_agreement":null},{"id":"W2763406350","doi":"10.1017/s1351324917000389","title":"Emerging trends: A tribute to Charles Wayne","year":2017,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alberta-Pacific Forest Industries","keywords":"Tribute; Order (exchange); Government (linguistics); Computer science; sort; Management; Law; Political science; Economics; Finance; Philosophy; Linguistics","score_opus":0.007743718129550689,"score_gpt":0.2627037147876462,"score_spread":0.2549599966580955,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2763406350","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00019490285,0.0067819655,0.0003568605,0.96973026,0.01812965,0.0000036173444,0.000024968795,0.00003502002,0.004742795],"genre_scores_gemma":[0.020199949,0.020470206,0.0017356803,0.83595127,0.047266953,0.000061929946,0.00010024365,0.00047202577,0.07374173],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98769814,0.0030154504,0.00075471,0.0025050486,0.0048552626,0.0011713794],"domain_scores_gemma":[0.93989766,0.02688215,0.0016461777,0.002938163,0.016641326,0.011994595],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01575915,0.0008568791,0.0012961156,0.0029757172,0.0073018204,0.01879287,0.002249162,0.013699069,0.013052971],"category_scores_gemma":[0.061651226,0.00056502223,0.00067744823,0.002038947,0.015903145,0.020904101,0.005554382,0.036342714,0.007030458],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000075341695,0.000012684683,0.0001305125,0.000025320795,0.000007557543,0.00008738761,0.0005173594,0.00003871291,0.00004627972,0.0289161,0.9585378,0.011672742],"study_design_scores_gemma":[0.0000053023687,0.0000066878556,0.00010618967,0.00020265137,0.0000029021987,0.000084668005,0.0006347001,0.000057756683,0.00006543404,0.01618713,0.9826272,0.00001930467],"about_ca_topic_score_codex":0.011673778,"about_ca_topic_score_gemma":0.015602261,"teacher_disagreement_score":0.01879287,"about_ca_system_score_codex":0.0070314147,"about_ca_system_score_gemma":0.010078143,"threshold_uncertainty_score":0.08334339},"labels":[],"label_agreement":null},{"id":"W2946534924","doi":"10.1017/s1351324919000135","title":"NLP commercialisation in the last 25 years","year":2019,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Focus (optics); Quarter (Canadian coin); Point (geometry); Artificial intelligence; Natural (archaeology); Natural language processing; History","score_opus":0.004465130453185606,"score_gpt":0.23091334490317533,"score_spread":0.22644821444998972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946534924","genre_codex":"other","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09056501,0.28123158,0.019730018,0.24481165,0.030755844,0.00026436866,0.0043603573,0.0039471723,0.32433403],"genre_scores_gemma":[0.5040691,0.17101409,0.043579295,0.052458603,0.01940721,0.00040454612,0.013507564,0.0044859066,0.19107372],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9913591,0.0014154625,0.0006569955,0.0014739536,0.0044990126,0.0005955768],"domain_scores_gemma":[0.94728225,0.01798301,0.00483877,0.003242474,0.021938076,0.004715478],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017618125,0.00053683383,0.0005158945,0.004283735,0.0018050934,0.012217952,0.0016752691,0.0032287717,0.017646514],"category_scores_gemma":[0.041836906,0.00040417176,0.0006015472,0.0052010226,0.0035602094,0.011602894,0.004659221,0.003371062,0.005631595],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034321073,0.00018818326,0.003363188,0.0018992937,0.00006387653,0.0011780526,0.0018700064,0.0007152286,0.005005743,0.13024175,0.29538655,0.55974483],"study_design_scores_gemma":[0.000012347139,0.00004129281,0.003075311,0.0005107394,0.00001249894,0.00043042089,0.0006200393,0.0006243361,0.0015141311,0.006982847,0.9861501,0.000025917629],"about_ca_topic_score_codex":0.0025811594,"about_ca_topic_score_gemma":0.0022252537,"teacher_disagreement_score":0.017646514,"about_ca_system_score_codex":0.005496271,"about_ca_system_score_gemma":0.004207455,"threshold_uncertainty_score":0.09317464},"labels":[],"label_agreement":null},{"id":"W3009120927","doi":"10.1017/s1351324920000133","title":"Nonuniform language in technical writing: Detection and correction","year":2020,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Text Readability and Simplification","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Natural language processing; Readability; Paraphrase; Artificial intelligence; Task (project management); Sentence; Context (archaeology); Synonym (taxonomy); Similarity (geometry); Programming language","score_opus":0.004691055894078967,"score_gpt":0.21110933476405405,"score_spread":0.20641827886997507,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3009120927","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43956098,0.0011795298,0.5450733,0.0005878599,0.00024210426,0.00035335598,0.00076448155,0.008713036,0.003525352],"genre_scores_gemma":[0.72165173,0.0002507121,0.2740336,0.00011326148,0.00008261161,0.00009201342,0.0008960045,0.00038159534,0.0024985368],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99435896,0.001703235,0.0006724451,0.0013671541,0.0017064754,0.00019170898],"domain_scores_gemma":[0.96519655,0.014177757,0.006305132,0.0047214525,0.00898364,0.0006154729],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025170567,0.0007549339,0.0007767307,0.003692467,0.00056969444,0.0014435868,0.0013581086,0.00091983884,0.0018724914],"category_scores_gemma":[0.021265794,0.00027546787,0.00042578919,0.0018717733,0.0007842088,0.001401531,0.001260595,0.00077972485,0.001368931],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053932494,0.00023726732,0.036393188,0.0008363486,0.00011268428,0.0010830136,0.0012383864,0.0071710497,0.13436453,0.0014231899,0.0059363977,0.8106646],"study_design_scores_gemma":[0.00008156999,0.00042775052,0.065689266,0.00014944655,0.00015642653,0.003954249,0.0013959933,0.5438369,0.36401117,0.0036598316,0.01646955,0.00016780403],"about_ca_topic_score_codex":0.0014941442,"about_ca_topic_score_gemma":0.0023927733,"teacher_disagreement_score":0.003692467,"about_ca_system_score_codex":0.0004416045,"about_ca_system_score_gemma":0.00076223584,"threshold_uncertainty_score":0.0133116245},"labels":[],"label_agreement":null},{"id":"W3042081424","doi":"10.1017/s1351324920000376","title":"Negation detection for sentiment analysis: A case study in Spanish","year":2020,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Negation; Computer science; Natural language processing; Sentiment analysis; Scope (computer science); Artificial intelligence; Identification (biology); Task (project management); Context (archaeology); Programming language","score_opus":0.011398076280530871,"score_gpt":0.25877069535249575,"score_spread":0.24737261907196487,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3042081424","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94309145,0.0010525736,0.027733028,0.0042672916,0.00020742853,0.0003392085,0.0012005136,0.00066490564,0.021443503],"genre_scores_gemma":[0.96353126,0.0007975101,0.02725538,0.00071977364,0.00008423092,0.000104448365,0.0010894829,0.00033640963,0.006081631],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9981489,0.000820623,0.00013198175,0.00019121288,0.0005768647,0.00013041873],"domain_scores_gemma":[0.9907273,0.004992947,0.0004519163,0.00043461277,0.0031091478,0.00028404046],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033138602,0.00051362324,0.00034553837,0.0011342084,0.0012312082,0.0013719591,0.0006932235,0.0011349849,0.001696068],"category_scores_gemma":[0.0149469515,0.0001331724,0.0003641557,0.0013269136,0.0008062138,0.00076269824,0.00086754985,0.00071048114,0.0006390769],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013733093,0.0017724575,0.22924161,0.0021496569,0.0001931768,0.0669272,0.05179111,0.013179717,0.047789223,0.011015917,0.062013127,0.5125535],"study_design_scores_gemma":[0.00047884436,0.00091173366,0.2258507,0.0011563213,0.0003529171,0.027738744,0.08349016,0.1435168,0.07639556,0.017812755,0.42202115,0.00027439956],"about_ca_topic_score_codex":0.02677096,"about_ca_topic_score_gemma":0.02742238,"teacher_disagreement_score":0.02677096,"about_ca_system_score_codex":0.0017864378,"about_ca_system_score_gemma":0.0013020668,"threshold_uncertainty_score":0.053230226},"labels":[],"label_agreement":null},{"id":"W3044693911","doi":"10.1017/s1351324920000509","title":"Comparison of rule-based and neural network models for negation detection in radiology reports","year":2020,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institute of Population and Public Health","funders":"Engineering and Physical Sciences Research Council; Medical Research Council","keywords":"Computer science; Artificial intelligence; Natural language processing; Negation; Artificial neural network; Sentence; Syntax; Pipeline (software); Machine learning; Rule-based system; Python (programming language); Information retrieval; Programming language","score_opus":0.014563883407551955,"score_gpt":0.24777112941400728,"score_spread":0.23320724600645532,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3044693911","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5326376,0.0058492064,0.43569937,0.0036084526,0.00057265867,0.0005153365,0.0034192207,0.0077259648,0.009972169],"genre_scores_gemma":[0.910329,0.0006360064,0.08349852,0.0005368945,0.000119071025,0.00017253683,0.0024547828,0.00012280505,0.002130312],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989027,0.00036643056,0.00011679511,0.00030582887,0.000228522,0.00007973779],"domain_scores_gemma":[0.9859061,0.011605394,0.00058524875,0.00031284848,0.0014164426,0.00017382925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004678042,0.0012416691,0.0010504443,0.0016804346,0.000432223,0.0019191223,0.0023856722,0.0015261213,0.0027010208],"category_scores_gemma":[0.0154606495,0.0005410477,0.0011469333,0.0008393092,0.0004920559,0.0022790572,0.0007918771,0.0017399978,0.0009059091],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009978489,0.00044433514,0.009266322,0.00026931326,0.00031936204,0.00023350949,0.00011754007,0.84079385,0.0014755734,0.0020242713,0.003461715,0.14059636],"study_design_scores_gemma":[0.000009984504,0.00002563378,0.0002708659,0.0000123686095,0.000016785634,0.000011251044,0.000005782727,0.998372,0.00028805103,0.0008799936,0.000102265294,0.0000050490685],"about_ca_topic_score_codex":0.018603964,"about_ca_topic_score_gemma":0.013474505,"teacher_disagreement_score":0.018603964,"about_ca_system_score_codex":0.0018728069,"about_ca_system_score_gemma":0.0012761076,"threshold_uncertainty_score":0.0369913},"labels":[],"label_agreement":null},{"id":"W4226024751","doi":"10.1017/s1351324922000134","title":"Real-world sentence boundary detection using multitask learning: A case study on French","year":2022,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Hanbat National University","keywords":"Computer science; Sentence; Punctuation; Natural language processing; Task (project management); Artificial intelligence; Boundary (topology); Multi-task learning; Speech recognition","score_opus":0.01083990589589887,"score_gpt":0.27614783946117666,"score_spread":0.26530793356527776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226024751","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9574623,0.001284699,0.02884295,0.001378357,0.0001512362,0.00023538154,0.0038111156,0.002708739,0.0041251453],"genre_scores_gemma":[0.95157665,0.00023939359,0.039300818,0.00037780104,0.00010288897,0.00015068515,0.0059284847,0.0001912287,0.002132082],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99726033,0.0015348417,0.00015385561,0.00044925243,0.00043784807,0.00016385686],"domain_scores_gemma":[0.98725617,0.008560916,0.00061626226,0.0009797586,0.0020820107,0.00050492096],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002485326,0.0008659191,0.00051600044,0.0019204443,0.0015498062,0.0013158513,0.0011014926,0.0019108533,0.001264634],"category_scores_gemma":[0.012158814,0.00015175942,0.00056169444,0.0017077023,0.00075050996,0.0013251612,0.00072147825,0.0007789317,0.00074639946],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033407833,0.003347955,0.11413224,0.0037814865,0.00070394657,0.03303708,0.01725966,0.06382222,0.08793525,0.006566803,0.09650704,0.5695656],"study_design_scores_gemma":[0.00061591616,0.002433633,0.23996113,0.0004910034,0.00044454282,0.012296382,0.019767428,0.41209218,0.12369832,0.013379518,0.17427783,0.0005421676],"about_ca_topic_score_codex":0.031283077,"about_ca_topic_score_gemma":0.043383162,"teacher_disagreement_score":0.031283077,"about_ca_system_score_codex":0.0012695241,"about_ca_system_score_gemma":0.0007627204,"threshold_uncertainty_score":0.062201977},"labels":[],"label_agreement":null},{"id":"W4284891255","doi":"10.1017/s1351324922000298","title":"Neural automated writing evaluation for Korean L2 writing","year":2022,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"National Research Foundation of Korea; National Research Foundation","keywords":"Computer science; Fluency; Parsing; Artificial intelligence; Natural language processing; Complement (music); Artificial neural network; Reliability (semiconductor); Linguistics","score_opus":0.009429505732520903,"score_gpt":0.2826202360865886,"score_spread":0.2731907303540677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4284891255","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85569,0.0004251754,0.13131659,0.00021559361,0.00012646602,0.00021881271,0.00076836755,0.006594815,0.0046441066],"genre_scores_gemma":[0.95234835,0.00008083415,0.04262829,0.000064550186,0.000013855467,0.00009429042,0.001001994,0.00007833339,0.0036895254],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989942,0.0003936054,0.000086775726,0.00024162781,0.00021965493,0.000064189095],"domain_scores_gemma":[0.99698323,0.0012903925,0.00019924615,0.00035344635,0.0010623903,0.000111369445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017273132,0.0005999577,0.00047576678,0.00075606746,0.00029553022,0.0008095186,0.00059963844,0.0005294825,0.0030766595],"category_scores_gemma":[0.0063184286,0.00016785125,0.0002603517,0.00041797885,0.00021497361,0.0012159623,0.00087762426,0.0006184797,0.0009598836],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085651595,0.00076151994,0.014976845,0.00028290268,0.00013653561,0.00036187185,0.00036682075,0.039574616,0.05595846,0.0008169355,0.0050155595,0.8808915],"study_design_scores_gemma":[0.000046885503,0.0003766963,0.013788265,0.00002541867,0.00003811423,0.0001306478,0.00022279918,0.936131,0.046965178,0.00057712325,0.0016618242,0.000036049092],"about_ca_topic_score_codex":0.0032810208,"about_ca_topic_score_gemma":0.0046982537,"teacher_disagreement_score":0.0032810208,"about_ca_system_score_codex":0.00052044314,"about_ca_system_score_gemma":0.00042480358,"threshold_uncertainty_score":0.01029247},"labels":[],"label_agreement":null},{"id":"W4382542296","doi":"10.1017/s1351324923000311","title":"Korean named entity recognition based on language-specific features","year":2023,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Morpheme; Natural language processing; Artificial intelligence; Annotation; Named-entity recognition; Scheme (mathematics); Ambiguity; Segmentation; Syllable; Word (group theory); Speech recognition; Task (project management); Linguistics","score_opus":0.010039043517492463,"score_gpt":0.22310503279439328,"score_spread":0.21306598927690082,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382542296","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15574741,0.0008731891,0.83165187,0.00027880032,0.00017315232,0.000106502834,0.0010783841,0.005788933,0.004301824],"genre_scores_gemma":[0.7500292,0.0005328913,0.24234214,0.00008902161,0.000046312773,0.00006321108,0.003723466,0.0001840015,0.0029897],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99938977,0.00016133722,0.00005693959,0.00023238032,0.00010510495,0.000054421802],"domain_scores_gemma":[0.99864703,0.0004862421,0.00012377591,0.00032964445,0.00036990427,0.000043355478],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010613514,0.00064784486,0.00066586316,0.0012070679,0.00029259242,0.0010093523,0.00093138206,0.00051834347,0.0016174128],"category_scores_gemma":[0.002474282,0.00018835065,0.00088426546,0.0014801233,0.00025356896,0.003891006,0.00066732406,0.0006798471,0.001289908],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005904056,0.00024981756,0.01550482,0.00033448325,0.00036559647,0.00045071755,0.0002859675,0.15452266,0.06398987,0.008916793,0.0068262178,0.7479627],"study_design_scores_gemma":[0.0000071497557,0.000052760563,0.0040274127,0.00001622231,0.00009007432,0.00013783111,0.00007113719,0.964128,0.026018213,0.0017805266,0.0036344968,0.000036160032],"about_ca_topic_score_codex":0.0035173649,"about_ca_topic_score_gemma":0.0050680996,"teacher_disagreement_score":0.0035173649,"about_ca_system_score_codex":0.00032245374,"about_ca_system_score_gemma":0.000385711,"threshold_uncertainty_score":0.0069937706},"labels":[],"label_agreement":null},{"id":"W4384694635","doi":"10.1017/s1351324923000360","title":"Describe the house and I will tell you the price: House price prediction with textual description data","year":2023,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Housing Market and Economics","field":"Economics, Econometrics and Finance","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Word2vec; Computer science; House price; Boosting (machine learning); Word embedding; Artificial intelligence; Word (group theory); Machine learning; Information retrieval; Data mining; Natural language processing; Embedding; Econometrics; Linguistics; Mathematics","score_opus":0.019698140741209,"score_gpt":0.19196146109524972,"score_spread":0.17226332035404074,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384694635","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9694806,0.00041318426,0.017398668,0.00069750083,0.00013024053,0.000043315275,0.008567474,0.00089833245,0.0023707475],"genre_scores_gemma":[0.9665816,0.00016425928,0.016808853,0.000094611314,0.00007013551,0.00003153685,0.013631633,0.00002960698,0.002587766],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99975723,0.00008138671,0.000019122448,0.000054328484,0.00006127195,0.000026760588],"domain_scores_gemma":[0.99842334,0.00096876867,0.00018845034,0.000112213136,0.0002426933,0.0000645632],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048020086,0.0005596789,0.00022480465,0.00081931666,0.00012480227,0.00042103734,0.0003831632,0.00056911213,0.002966457],"category_scores_gemma":[0.0031204163,0.0001449446,0.0003453452,0.0010459616,0.00014136013,0.0009412819,0.00024432573,0.0007261248,0.0014342716],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025236094,0.0021448736,0.29265153,0.0006068831,0.0003364012,0.0014505822,0.00029702936,0.1994756,0.013320161,0.0023297044,0.0563114,0.4285522],"study_design_scores_gemma":[0.000023769659,0.00009937906,0.040789768,0.000021680778,0.000031057647,0.000103182676,0.000110460416,0.95171463,0.0039157397,0.0009251154,0.0022461864,0.000018882218],"about_ca_topic_score_codex":0.011261272,"about_ca_topic_score_gemma":0.014413327,"teacher_disagreement_score":0.011261272,"about_ca_system_score_codex":0.00041039367,"about_ca_system_score_gemma":0.00017410552,"threshold_uncertainty_score":0.022391498},"labels":[],"label_agreement":null},{"id":"W4385552004","doi":"10.1017/s1351324923000396","title":"SSL-GAN-RoBERTa: A robust semi-supervised model for detecting Anti-Asian COVID-19 hate speech on social media","year":2023,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Social media; Transformer; Artificial intelligence; Task (project management); Coronavirus disease 2019 (COVID-19); Machine learning; Speech recognition; Data mining; Voltage","score_opus":0.025940205091409797,"score_gpt":0.26037499805348296,"score_spread":0.23443479296207317,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385552004","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12166406,0.0027926122,0.85606223,0.001752146,0.00040222373,0.0004246563,0.0020269367,0.009575964,0.0052991556],"genre_scores_gemma":[0.8490756,0.00053716084,0.13234217,0.001404941,0.00032666774,0.0005622672,0.0054491684,0.00037684359,0.009925169],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99875915,0.00050467654,0.000048994738,0.0004282052,0.00014273172,0.00011612288],"domain_scores_gemma":[0.9975339,0.0014135711,0.00020391727,0.00024639285,0.0004856906,0.00011648056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023119564,0.0017194476,0.001717782,0.0008338332,0.00045560702,0.00084623107,0.0031591903,0.0017447409,0.0014556132],"category_scores_gemma":[0.0041897795,0.0006216286,0.0013876349,0.000531763,0.0010290974,0.0013713561,0.0010852882,0.0027256373,0.0013088208],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006537822,0.0006043523,0.005909202,0.0002950033,0.00025689695,0.00028010478,0.00022204801,0.7308505,0.007111074,0.0038897954,0.019588271,0.23033898],"study_design_scores_gemma":[0.000007908388,0.000027695016,0.0001835398,0.0000061547616,0.000007848043,0.000015670697,0.0000070232363,0.9980325,0.00051124254,0.00091537833,0.00027830616,0.000006784386],"about_ca_topic_score_codex":0.008012514,"about_ca_topic_score_gemma":0.011673475,"teacher_disagreement_score":0.008012514,"about_ca_system_score_codex":0.0010410631,"about_ca_system_score_gemma":0.0012642792,"threshold_uncertainty_score":0.015931726},"labels":[],"label_agreement":null},{"id":"W4388731527","doi":"10.1017/s1351324923000529","title":"Korean named entity recognition based on language-specific features – CORRIGENDUM","year":2023,"lang":"en","type":"erratum","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"National Research Foundation of Korea; National Research Foundation","keywords":"Computer science; Content (measure theory); Natural language processing; Information retrieval; Artificial intelligence; Database; World Wide Web","score_opus":0.012516482264891337,"score_gpt":0.24193848754718925,"score_spread":0.22942200528229792,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388731527","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00818963,0.00768495,0.066276655,0.041893438,0.77666897,0.0003066548,0.015267618,0.0085116085,0.075200476],"genre_scores_gemma":[0.049323853,0.0108989095,0.07350224,0.017696485,0.028717458,0.00024474287,0.040096596,0.005645645,0.77387404],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99922514,0.00010362883,0.00014809234,0.00014157427,0.00033309375,0.00004859775],"domain_scores_gemma":[0.9948537,0.0005056783,0.00014649011,0.00063057843,0.0037378373,0.0001258131],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008945065,0.0010885131,0.0008209739,0.0015389748,0.00094739604,0.002120643,0.0012801582,0.0010731411,0.07033081],"category_scores_gemma":[0.007221515,0.00042499238,0.0006416319,0.0017452494,0.000541109,0.0018039905,0.00092916755,0.0014088188,0.071740635],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048113256,0.000020030937,0.00016565746,0.00014720493,0.000016505339,0.0006597371,0.000025367883,0.00018243308,0.0008989096,0.0012556391,0.9561004,0.040479936],"study_design_scores_gemma":[0.000018909435,0.000037583457,0.0021162229,0.00014194295,0.000055041583,0.0011082622,0.00012619454,0.0023601633,0.005663846,0.0025396468,0.9857772,0.000055014927],"about_ca_topic_score_codex":0.012396283,"about_ca_topic_score_gemma":0.019838339,"teacher_disagreement_score":0.07033081,"about_ca_system_score_codex":0.0010301945,"about_ca_system_score_gemma":0.0010784047,"threshold_uncertainty_score":0.23528004},"labels":[],"label_agreement":null},{"id":"W4390789641","doi":"10.1017/s1351324923000542","title":"Lightweight transformers for clinical natural language processing","year":2024,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"NIHR Imperial Biomedical Research Centre; Instituto de Salud Carlos III; Medical Research Council; National Institutes of Health; Kementerian Kesihatan Malaysia; All-India Institute of Medical Sciences; Horizon 2020 Framework Programme; Prince Charles Hospital Foundation; Foreign, Commonwealth and Development Office; National Institute for Health Research Health Protection Research Unit; University of Oxford; Norges Forskningsråd; Imperial College London; Conselho Nacional de Desenvolvimento Científico e Tecnológico; University of Cape Town; Public Health England; Wellcome Trust; University College Dublin; Sunnybrook Research Institute; Institut National de la Santé et de la Recherche Médicale; Canadian Institutes of Health Research; National Institute for Health and Care Research; Ministero della Salute; European Federation of Pharmaceutical Industries and Associations; European Commission; Bill and Melinda Gates Foundation","keywords":"Computer science; Transformer; Natural language processing; Artificial intelligence; Electrical engineering","score_opus":0.011690292682715328,"score_gpt":0.30424274043960325,"score_spread":0.29255244775688793,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390789641","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01791121,0.0013522463,0.8986432,0.0024453534,0.00032328002,0.0005248201,0.0059841727,0.06904554,0.003770186],"genre_scores_gemma":[0.37173125,0.0012735588,0.59954345,0.0014416703,0.0002141247,0.0006950929,0.016596658,0.002284962,0.0062192585],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99769956,0.0007538392,0.00026675235,0.0005486269,0.0006125403,0.000118684984],"domain_scores_gemma":[0.99324965,0.0037991249,0.00034989297,0.0016103536,0.00077499886,0.00021602071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025475745,0.0013141092,0.000578846,0.0016842915,0.00034939847,0.00183358,0.002084572,0.00090729375,0.013515056],"category_scores_gemma":[0.015458226,0.00071737455,0.0013196762,0.0012548575,0.0010818698,0.0061303945,0.0037677381,0.0031775904,0.008818846],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010626093,0.0004112517,0.0043865633,0.0010206095,0.00017790166,0.0005135506,0.00048023282,0.1242555,0.01213633,0.055408567,0.06832554,0.73182124],"study_design_scores_gemma":[0.00012118371,0.0002119466,0.00047860344,0.0001226599,0.00006660144,0.00036045216,0.00014898332,0.8164868,0.015935369,0.12368374,0.042323884,0.000059856327],"about_ca_topic_score_codex":0.0031337882,"about_ca_topic_score_gemma":0.00571743,"teacher_disagreement_score":0.013515056,"about_ca_system_score_codex":0.0015541998,"about_ca_system_score_gemma":0.0029153815,"threshold_uncertainty_score":0.045212388},"labels":[],"label_agreement":null}]}