{"meta":{"query_hash":"156f0c5b8dc6","filters":{"venue":"International Conference on Computational Linguistics"},"cohort_total":34,"direct_labels_cover":0,"predictions_cover":34,"exported":34,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/156f0c5b8dc6","api":"https://metacan.xera.ac/api/v1/cohort?venue=International+Conference+on+Computational+Linguistics"},"results":[{"id":"W169800830","doi":"","title":"Scaling up Analogical Learning","year":2008,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Scalability; Artificial intelligence; Simple (philosophy); Scaling; Identification (biology); Space (punctuation); Limit (mathematics); Machine learning; Theoretical computer science; Mathematics; Epistemology","score_opus":0.06614344725555805,"score_gpt":0.337967393007861,"score_spread":0.27182394575230295,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W169800830","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2751903,0.009223788,0.63613653,0.007733268,0.0010775697,0.0007592945,0.0011132215,0.015299004,0.053467005],"genre_scores_gemma":[0.72387534,0.002400458,0.2629527,0.001203012,0.0004132334,0.000544758,0.0011050175,0.00090571726,0.006599781],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961792,0.0010831343,0.00026721242,0.0009207384,0.0012891871,0.0002605345],"domain_scores_gemma":[0.9659475,0.021247124,0.00069534744,0.008539135,0.0029272821,0.0006436312],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032005082,0.0015916957,0.0020583135,0.0017078992,0.0008928993,0.0027442572,0.0040817168,0.0021640963,0.028981091],"category_scores_gemma":[0.04074112,0.000781142,0.0010842269,0.0030743014,0.0016634499,0.01235488,0.0050320555,0.00309815,0.0057240343],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000652251,0.0011370883,0.0027634988,0.00091086596,0.00021478771,0.00020854342,0.00045769603,0.11147424,0.015895767,0.0581622,0.021370748,0.7867523],"study_design_scores_gemma":[0.0002721092,0.00032811536,0.00084049173,0.00006804761,0.00010451299,0.00025384492,0.00032493836,0.76836306,0.00873557,0.20636694,0.014298182,0.000044115877],"about_ca_topic_score_codex":0.0025916004,"about_ca_topic_score_gemma":0.0019553495,"teacher_disagreement_score":0.028981091,"about_ca_system_score_codex":0.0015210363,"about_ca_system_score_gemma":0.0014921302,"threshold_uncertainty_score":0.096951425},"labels":[],"label_agreement":null},{"id":"W1815301076","doi":"","title":"Measuring the Non-compositionality of Multiword Expressions","year":2010,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Principle of compositionality; Computer science; Artificial intelligence; Natural language processing; Semantics (computer science); Metric (unit); Natural language; Question answering; Combinatory categorial grammar; Distributional semantics; Expression (computer science); Information extraction; Programming language; Semantic similarity; Link grammar; Rule-based machine translation","score_opus":0.0730969691908334,"score_gpt":0.32514661388054306,"score_spread":0.25204964468970964,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1815301076","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4840883,0.0015936336,0.5077538,0.00024741516,0.00015708826,0.00017966494,0.0005141906,0.0011574134,0.0043085963],"genre_scores_gemma":[0.8126592,0.0005931136,0.18261057,0.00008799208,0.00009338513,0.00016966919,0.0018266233,0.0003244131,0.001634966],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9962781,0.001206149,0.0005335329,0.00095445727,0.00087545,0.00015233975],"domain_scores_gemma":[0.98637044,0.008462582,0.0015140852,0.0014831077,0.0018333149,0.00033644974],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024562236,0.00074496906,0.0007226532,0.0028273195,0.0008347637,0.001822397,0.0008609626,0.0011275944,0.0013259294],"category_scores_gemma":[0.020456329,0.00041790915,0.00062590925,0.0019135417,0.00088648585,0.0048124897,0.0026679682,0.0010338421,0.0010175563],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00095899747,0.000299506,0.052718654,0.001413574,0.00039297243,0.0006661928,0.0022997411,0.012379215,0.29919678,0.016817246,0.001576804,0.61128026],"study_design_scores_gemma":[0.0000940984,0.0012326004,0.10833195,0.00023426262,0.00049042504,0.0036937569,0.0033801463,0.5285692,0.24974047,0.077282645,0.02670508,0.00024537242],"about_ca_topic_score_codex":0.00048230393,"about_ca_topic_score_gemma":0.00083859364,"teacher_disagreement_score":0.0028273195,"about_ca_system_score_codex":0.00038023363,"about_ca_system_score_gemma":0.00059572665,"threshold_uncertainty_score":0.012989879},"labels":[],"label_agreement":null},{"id":"W1956103381","doi":"","title":"Automatic Acquisition of Lexical Formality","year":2010,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Formality; Computer science; Word (group theory); Natural language processing; Artificial intelligence; Word Association; Metric (unit); Similarity (geometry); Association (psychology); Task (project management); Synonym (taxonomy); Linguistics; Speech recognition; Psychology","score_opus":0.0284122819659165,"score_gpt":0.3393629900576921,"score_spread":0.3109507080917756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1956103381","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31549624,0.0018253406,0.6265804,0.0007653699,0.00040077374,0.00057657313,0.0050962,0.032364693,0.0168944],"genre_scores_gemma":[0.66334444,0.00051767763,0.32234108,0.0001528582,0.0001473376,0.00027219663,0.008415918,0.0012969736,0.0035115369],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976675,0.000461404,0.00030264448,0.00085575506,0.0005591194,0.00015351648],"domain_scores_gemma":[0.9901086,0.004382833,0.0010587602,0.0014617491,0.0026734597,0.0003145325],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021890511,0.0009597391,0.0012739331,0.0072820806,0.00080276985,0.0029582642,0.001291727,0.00061217614,0.0060697906],"category_scores_gemma":[0.015714016,0.0008203471,0.00070406037,0.002430234,0.00077885936,0.005937049,0.00400361,0.0014518676,0.0038987005],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029409584,0.00014376038,0.018412687,0.00089042174,0.00010956997,0.00055050413,0.0015796226,0.0014727585,0.115010835,0.011439104,0.009114193,0.84098244],"study_design_scores_gemma":[0.00034373943,0.0007875534,0.11593482,0.00083261187,0.00048638546,0.006344464,0.005297728,0.41881517,0.18925267,0.1408122,0.12052107,0.0005716207],"about_ca_topic_score_codex":0.0011920779,"about_ca_topic_score_gemma":0.0021416396,"teacher_disagreement_score":0.0072820806,"about_ca_system_score_codex":0.0006262375,"about_ca_system_score_gemma":0.0014796808,"threshold_uncertainty_score":0.020305455},"labels":[],"label_agreement":null},{"id":"W2108711363","doi":"","title":"A Strategy of Mapping Polish WordNet onto Princeton WordNet","year":2012,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Premise; Computer science; Set (abstract data type); Natural language processing; Focus (optics); Artificial intelligence; Lexical database; Range (aeronautics); Information retrieval; Linguistics; Programming language; Philosophy","score_opus":0.061792355341779494,"score_gpt":0.3410952297029126,"score_spread":0.2793028743611331,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2108711363","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020821882,0.00008530088,0.95314974,0.0007366914,0.00017438557,0.00054122874,0.00086886826,0.002520006,0.02110184],"genre_scores_gemma":[0.15012999,0.00028344494,0.8307717,0.00038639936,0.000046324847,0.0015656503,0.0020446081,0.0009836874,0.013788344],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99897134,0.00031907775,0.00010487304,0.00031504204,0.00021880194,0.00007083003],"domain_scores_gemma":[0.9986494,0.0002553635,0.00006167174,0.0006309948,0.0003364067,0.00006613094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015546201,0.00074456417,0.00041947895,0.0038048774,0.001896846,0.0021924034,0.00095525663,0.0005701733,0.007832087],"category_scores_gemma":[0.0066505377,0.000815952,0.0007725426,0.0027350853,0.0012703557,0.0054285754,0.0043884846,0.0016520127,0.0034176542],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001356405,0.0001806922,0.0028933685,0.00026231856,0.000091639544,0.0004149135,0.002980398,0.004939711,0.014568421,0.57188624,0.012959554,0.3886872],"study_design_scores_gemma":[0.00006288382,0.0002736166,0.003636281,0.00017999866,0.000112112044,0.00086935185,0.0025020281,0.066011176,0.047235698,0.57558644,0.3033997,0.00013072084],"about_ca_topic_score_codex":0.004162129,"about_ca_topic_score_gemma":0.0061406265,"teacher_disagreement_score":0.007832087,"about_ca_system_score_codex":0.0007372301,"about_ca_system_score_gemma":0.0017060741,"threshold_uncertainty_score":0.02620089},"labels":[],"label_agreement":null},{"id":"W2250555305","doi":"","title":"On Panini and the Generative Capacity of Contextualized Replacement Systems","year":2012,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"semigroups and automata theory","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Generative grammar; Formalism (music); Sanskrit; Computer science; Grammar; Rewriting; Linguistics; Mildly context-sensitive grammar formalism; Adaptive grammar; Emergent grammar; Natural language processing; Programming language; Mathematics; Artificial intelligence; Philosophy; Literature","score_opus":0.0576393092193052,"score_gpt":0.3040569194395782,"score_spread":0.24641761022027298,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250555305","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18101883,0.009895286,0.418433,0.0118607255,0.00030506204,0.00010005495,0.00028881812,0.001163764,0.3769344],"genre_scores_gemma":[0.957053,0.0016270387,0.030915545,0.00051056425,0.00022170803,0.00006525363,0.0001187373,0.00024627824,0.009241823],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99815565,0.0009180547,0.00009165303,0.00037254902,0.00026723644,0.00019485893],"domain_scores_gemma":[0.99333584,0.0050632465,0.00024203025,0.0009512084,0.00026441625,0.00014322542],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033299648,0.00043638019,0.00059623644,0.001246851,0.001790991,0.0029051213,0.000925817,0.0010897544,0.0068511353],"category_scores_gemma":[0.008463288,0.0005263391,0.00079232955,0.0011026915,0.011467322,0.0076781223,0.0034558561,0.0024184645,0.00065408694],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000061161295,0.0000022625948,0.00009789485,0.000012736669,0.0000014974033,0.00003437573,0.0006199578,0.00094407273,0.0001159211,0.99521494,0.0001450193,0.0028052053],"study_design_scores_gemma":[0.000004595683,0.000007798799,0.00007862467,0.000017510038,0.0000032970713,0.00005562307,0.0001171713,0.003432595,0.0002509713,0.98856527,0.007458364,0.000008149916],"about_ca_topic_score_codex":0.002062083,"about_ca_topic_score_gemma":0.0017655585,"teacher_disagreement_score":0.0068511353,"about_ca_system_score_codex":0.0018133362,"about_ca_system_score_gemma":0.00076681987,"threshold_uncertainty_score":0.022919357},"labels":[],"label_agreement":null},{"id":"W2250571015","doi":"","title":"Towards Automatic Topical Question Generation","year":2012,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Computer science","score_opus":0.09053207617422618,"score_gpt":0.3518676068766896,"score_spread":0.2613355307024634,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250571015","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023131201,0.00077882735,0.9401515,0.001647788,0.00073752203,0.0004541569,0.002308735,0.023192486,0.007597761],"genre_scores_gemma":[0.21969523,0.00046158046,0.7528078,0.000531251,0.0005123971,0.00052931416,0.013795103,0.0019235619,0.009743728],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967039,0.0014957131,0.0002412061,0.00076310063,0.0005378001,0.0002583547],"domain_scores_gemma":[0.9938467,0.0029065094,0.00016948419,0.0008881327,0.0019609334,0.00022824388],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034923558,0.0012508513,0.001346172,0.003915186,0.0015461795,0.0035698882,0.0017114736,0.0022302617,0.017081946],"category_scores_gemma":[0.010108998,0.0009436179,0.0016439509,0.001714645,0.0007190648,0.0047794003,0.003882206,0.0032538164,0.012831376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004678308,0.000340961,0.0020822452,0.000614998,0.000114529605,0.0003783944,0.0010807791,0.0067855907,0.0718988,0.036154035,0.080779396,0.7993025],"study_design_scores_gemma":[0.00014868853,0.00017336647,0.0014616614,0.000121853096,0.00017730739,0.00053162826,0.000894799,0.74154335,0.07270249,0.09500056,0.08717552,0.000068821064],"about_ca_topic_score_codex":0.0014722939,"about_ca_topic_score_gemma":0.0021231268,"teacher_disagreement_score":0.017081946,"about_ca_system_score_codex":0.00093499693,"about_ca_system_score_gemma":0.001736217,"threshold_uncertainty_score":0.05714482},"labels":[],"label_agreement":null},{"id":"W2251004534","doi":"","title":"Fourteen Light Tasks for comparing Analogical and Phrase-based Machine Translation","year":2014,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Transliteration; Machine translation; Phrase; Computer science; Scripting language; Natural language processing; Artificial intelligence; Rule-based machine translation; Translation (biology); Example-based machine translation; Machine translation software usability; Programming language","score_opus":0.058601602908245884,"score_gpt":0.33216916151273396,"score_spread":0.27356755860448806,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251004534","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84517705,0.015775105,0.07087881,0.0012048543,0.0026644939,0.0033554784,0.023417685,0.008212156,0.029314423],"genre_scores_gemma":[0.7610103,0.002359075,0.1320142,0.0012777715,0.0006137888,0.00367432,0.08832413,0.0013461462,0.009380348],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9933809,0.002359647,0.001229694,0.0013067678,0.0013801843,0.00034278154],"domain_scores_gemma":[0.98539484,0.009156359,0.00071274955,0.0023788724,0.0016592576,0.0006979499],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00574626,0.0021374046,0.0015439704,0.0040783104,0.0017554958,0.0020217367,0.0020260205,0.0025671537,0.0067741615],"category_scores_gemma":[0.020799503,0.0004760612,0.001618525,0.0030396278,0.0010875226,0.0029669208,0.0032732594,0.0020442468,0.003872226],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.014547103,0.005805582,0.031568866,0.008883675,0.002668289,0.0007248577,0.00091486744,0.032905865,0.041824576,0.005537467,0.074670896,0.779948],"study_design_scores_gemma":[0.008642872,0.021390041,0.25891095,0.0017492734,0.0041044247,0.004068796,0.0038429862,0.29032326,0.17043675,0.03642344,0.19893827,0.0011690015],"about_ca_topic_score_codex":0.0030277888,"about_ca_topic_score_gemma":0.005815649,"teacher_disagreement_score":0.0067741615,"about_ca_system_score_codex":0.0012638405,"about_ca_system_score_gemma":0.001475768,"threshold_uncertainty_score":0.030389488},"labels":[],"label_agreement":null},{"id":"W2251031731","doi":"","title":"A System for Multilingual Sentiment Learning On Large Data Sets","year":2012,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Sentiment analysis; Artificial intelligence; Generalization; Natural language processing; Set (abstract data type); Empirical research; Machine learning; Mathematics","score_opus":0.12093133476971414,"score_gpt":0.3898499392076454,"score_spread":0.26891860443793125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251031731","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07903781,0.00047014892,0.7322468,0.0012425137,0.00047611955,0.001371924,0.00760974,0.17055616,0.0069887266],"genre_scores_gemma":[0.15697362,0.00020408665,0.8238915,0.00043451812,0.00017651822,0.0008512212,0.010566245,0.0010057451,0.0058965827],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988078,0.0002683138,0.00015773295,0.00040737764,0.00028239656,0.00007631964],"domain_scores_gemma":[0.9970751,0.000998505,0.00020621074,0.00043144493,0.0011369141,0.00015181553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029487354,0.001304644,0.00090336625,0.002234167,0.0013758182,0.0013797343,0.0010654158,0.00095588894,0.008701602],"category_scores_gemma":[0.007303646,0.0005195507,0.0009276399,0.0017975474,0.0003030433,0.004067365,0.0020000928,0.0013587122,0.0071189376],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008020649,0.00042084392,0.006246318,0.0003947467,0.000229393,0.0003741218,0.0006288321,0.0028720335,0.054232996,0.003021212,0.06156396,0.86921346],"study_design_scores_gemma":[0.00029676215,0.0005559405,0.0101968525,0.00015843488,0.0002661491,0.0007444558,0.00086354354,0.808055,0.0795351,0.015449131,0.08369618,0.00018246945],"about_ca_topic_score_codex":0.0033380743,"about_ca_topic_score_gemma":0.0052548544,"teacher_disagreement_score":0.008701602,"about_ca_system_score_codex":0.0008092294,"about_ca_system_score_gemma":0.0009870073,"threshold_uncertainty_score":0.029109716},"labels":[],"label_agreement":null},{"id":"W2252045241","doi":"","title":"Flexible Structural Analysis of Near-Meet-Semilattices for Typed Unification-Based Grammar Design","year":2012,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Unification; Parsing; Computer science; Programming language; Head-driven phrase structure grammar; Grammar; Rule-based machine translation; Type (biology); Theoretical computer science; Algorithm; Natural language processing; Artificial intelligence; Generative grammar; Linguistics","score_opus":0.07928709696875265,"score_gpt":0.3722236565658698,"score_spread":0.2929365595971172,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2252045241","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005357228,0.00003409848,0.9925493,0.00008194221,0.000021967522,0.00004446135,0.00004784645,0.0005551083,0.0013081343],"genre_scores_gemma":[0.12889004,0.000074489675,0.8677093,0.0000639966,0.000032382606,0.00018862315,0.00026116104,0.00048847473,0.0022915725],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99708086,0.0008593098,0.0003103642,0.0003754041,0.0011751412,0.00019887532],"domain_scores_gemma":[0.9971706,0.0013128625,0.00019675161,0.0006031904,0.00055155705,0.0001650349],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035646579,0.0006699596,0.00063666404,0.0011983196,0.0013144398,0.002395982,0.0018613152,0.0008448368,0.0050290795],"category_scores_gemma":[0.0064921863,0.001015898,0.0018995444,0.0009352793,0.0024871188,0.0040143942,0.0025501992,0.002240656,0.0013641807],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009931222,0.00008744186,0.000889023,0.00020172048,0.000037770988,0.00040245088,0.001029485,0.06096716,0.017751954,0.8418993,0.0018212374,0.074813165],"study_design_scores_gemma":[0.000023872544,0.000040048857,0.00010241621,0.000046580328,0.000023167562,0.000119842356,0.00015932546,0.33300376,0.011209428,0.64597666,0.009261044,0.000033888366],"about_ca_topic_score_codex":0.0020105345,"about_ca_topic_score_gemma":0.003660196,"teacher_disagreement_score":0.0050290795,"about_ca_system_score_codex":0.0015543543,"about_ca_system_score_gemma":0.0023884007,"threshold_uncertainty_score":0.018851936},"labels":[],"label_agreement":null},{"id":"W2572452825","doi":"","title":"Selective Co-occurrences for Word-Emotion Association","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Word (group theory); Computer science; Association (psychology); Natural language processing; Word Association; Artificial intelligence; Task (project management); Emotion classification; Semantic similarity; Psychology; Linguistics","score_opus":0.05802061708972685,"score_gpt":0.35514580214277025,"score_spread":0.2971251850530434,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2572452825","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20722497,0.0011216436,0.7810143,0.00030619334,0.00018040436,0.0003273053,0.0016468927,0.0043106615,0.0038675084],"genre_scores_gemma":[0.6869684,0.00050257344,0.30145225,0.0001298968,0.00020384461,0.00061737053,0.006273242,0.00052315043,0.0033293131],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99828655,0.0003995385,0.00016314095,0.0007592672,0.00026350198,0.00012794044],"domain_scores_gemma":[0.9954513,0.0022633947,0.0005541928,0.00085824996,0.0006903378,0.00018259451],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014017279,0.0011320242,0.0010856941,0.003874016,0.0007703872,0.0010659498,0.0009654629,0.0008716413,0.002212543],"category_scores_gemma":[0.008338585,0.00039286222,0.0010626677,0.003845321,0.0007743163,0.0030649172,0.002146328,0.0015118719,0.0023064031],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009174516,0.00054486806,0.04328621,0.0005027491,0.00034205228,0.00026408673,0.0011132419,0.013343102,0.038434748,0.0050218888,0.008052931,0.88817674],"study_design_scores_gemma":[0.00008639182,0.00039840225,0.04320947,0.00009351523,0.0002173042,0.0010567806,0.0009764948,0.8788162,0.03489234,0.028416092,0.011714363,0.0001227022],"about_ca_topic_score_codex":0.0010896896,"about_ca_topic_score_gemma":0.0035490287,"teacher_disagreement_score":0.003874016,"about_ca_system_score_codex":0.00036824387,"about_ca_system_score_gemma":0.0009595037,"threshold_uncertainty_score":0.0074130893},"labels":[],"label_agreement":null},{"id":"W2573346038","doi":"","title":"Predicting sentential semantic compatibility for aggregation in text-to-text generation","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Sentence; Natural language processing; Artificial intelligence; Task (project management); Cluster analysis; Process (computing); Context (archaeology); Information retrieval","score_opus":0.07688502906661919,"score_gpt":0.3287613563504564,"score_spread":0.2518763272838372,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2573346038","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.63778776,0.001482883,0.33887175,0.0009890356,0.00028972852,0.00074766664,0.003938065,0.008942451,0.0069507193],"genre_scores_gemma":[0.81068647,0.00018119685,0.18186313,0.00007538965,0.000095577416,0.00021795338,0.0057386192,0.00033899726,0.0008026836],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99631566,0.0016773539,0.00033832705,0.00085574825,0.00059032295,0.0002225627],"domain_scores_gemma":[0.9759448,0.017852766,0.0013669856,0.0016041829,0.0027178419,0.00051334756],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056519513,0.0011222285,0.0007154646,0.003934626,0.0011307751,0.0019804933,0.0009407126,0.0013963438,0.001993907],"category_scores_gemma":[0.034592204,0.00037146654,0.0009722483,0.0023721163,0.00051181694,0.003942818,0.0014040117,0.0013635196,0.0012749054],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022870337,0.0010424894,0.077985264,0.0011454892,0.00039425585,0.000852232,0.0029683989,0.06360371,0.057282366,0.010073823,0.027827581,0.7545374],"study_design_scores_gemma":[0.00010224182,0.00036830318,0.036866017,0.00006849034,0.00019349463,0.00037216058,0.0008095256,0.89315146,0.044634435,0.01722133,0.006120567,0.00009199715],"about_ca_topic_score_codex":0.003557357,"about_ca_topic_score_gemma":0.0060920073,"teacher_disagreement_score":0.0056519513,"about_ca_system_score_codex":0.00091341796,"about_ca_system_score_gemma":0.001288601,"threshold_uncertainty_score":0.029890716},"labels":[],"label_agreement":null},{"id":"W2573843450","doi":"","title":"Determining the Multiword Expression Inventory of a Surprise Language","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Treebank; Surprise; Computer science; Natural language processing; Identification (biology); Artificial intelligence; Language model; Natural language; Language identification; Parsing; Psychology","score_opus":0.03742044249948087,"score_gpt":0.33147871213364416,"score_spread":0.29405826963416326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2573843450","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.80091864,0.0006407504,0.18112165,0.0004942706,0.00015683915,0.00011328266,0.0028509342,0.0028599547,0.010843648],"genre_scores_gemma":[0.91772896,0.00038561493,0.07061923,0.00013544512,0.00006157615,0.00012357272,0.006808242,0.00057130755,0.0035660733],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993356,0.00016302496,0.00006942808,0.00028134414,0.00008590304,0.00006463298],"domain_scores_gemma":[0.99782526,0.00092499296,0.00025732664,0.00024599748,0.000640376,0.00010607924],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010755688,0.0006038567,0.0004031326,0.0016019614,0.0004222373,0.0012200215,0.00046618594,0.00055127486,0.0032493253],"category_scores_gemma":[0.004425005,0.00055781094,0.00058325933,0.0006481379,0.0004697897,0.0031078944,0.0010855312,0.0013297905,0.0026374087],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009598696,0.00031213087,0.09591027,0.00087991747,0.00012635907,0.001935286,0.0051845545,0.0074650715,0.29911244,0.020927807,0.017647715,0.54953855],"study_design_scores_gemma":[0.000118129705,0.00083939516,0.20166504,0.00035674596,0.00033176076,0.006289718,0.008326554,0.5316431,0.1380048,0.03604274,0.076112986,0.00026906858],"about_ca_topic_score_codex":0.0012790109,"about_ca_topic_score_gemma":0.0016386814,"teacher_disagreement_score":0.0032493253,"about_ca_system_score_codex":0.0006053099,"about_ca_system_score_gemma":0.00080512214,"threshold_uncertainty_score":0.010870039},"labels":[],"label_agreement":null},{"id":"W2574914175","doi":"","title":"An interactive system for exploring community question answering forums","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Question answering; Computer science; World Wide Web; Interface (matter); Information retrieval; User interface; Graphical user interface; Human–computer interaction","score_opus":0.13824046408852536,"score_gpt":0.3595167125248636,"score_spread":0.22127624843633825,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2574914175","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035083596,0.00080519036,0.6966871,0.0005210279,0.00020263718,0.0011284478,0.009824833,0.24069145,0.015055738],"genre_scores_gemma":[0.20685954,0.00046488404,0.73389816,0.00040864918,0.0003004574,0.0028826725,0.03027757,0.008683364,0.016224628],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985973,0.00049616955,0.0000888418,0.00033512723,0.00036514679,0.00011751255],"domain_scores_gemma":[0.9946569,0.0034300706,0.00016950561,0.0005095545,0.0006763539,0.00055762863],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002468556,0.0013142282,0.0008115504,0.0034324701,0.0013826342,0.0019083596,0.0021866944,0.001540647,0.03620894],"category_scores_gemma":[0.008234875,0.0006387189,0.0007924497,0.0019283043,0.00044117324,0.004337752,0.004600656,0.00094814505,0.008040996],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002829979,0.0008219493,0.005865743,0.0019018976,0.0002307103,0.0012350305,0.00804736,0.004549758,0.068555154,0.016047858,0.19188261,0.69803196],"study_design_scores_gemma":[0.0013456183,0.0012022217,0.009638632,0.00041425615,0.00034152757,0.0020825847,0.0027726584,0.22402844,0.052937668,0.04358964,0.66110545,0.00054123456],"about_ca_topic_score_codex":0.001617231,"about_ca_topic_score_gemma":0.0022260745,"teacher_disagreement_score":0.03620894,"about_ca_system_score_codex":0.0005166664,"about_ca_system_score_gemma":0.00073869183,"threshold_uncertainty_score":0.12113094},"labels":[],"label_agreement":null},{"id":"W2574986170","doi":"","title":"plWordNet 3.0 - a Comprehensive Lexical-Semantic Resource.","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Computer science; Set (abstract data type); Natural language processing; Resource (disambiguation); Word (group theory); Artificial intelligence; Lexical database; Information retrieval; Word list; Linguistics; Class (philosophy); Programming language","score_opus":0.04651737362330788,"score_gpt":0.32945800977683776,"score_spread":0.2829406361535299,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2574986170","genre_codex":"dataset","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01812159,0.0043238923,0.37292784,0.0031946073,0.0015553829,0.0013566731,0.39583597,0.09180233,0.11088179],"genre_scores_gemma":[0.040277842,0.0031449562,0.19703236,0.0012388609,0.00026345393,0.002117562,0.6900859,0.023345282,0.04249373],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986473,0.00024207823,0.0002689118,0.00035198705,0.0003898349,0.00009986013],"domain_scores_gemma":[0.9979882,0.00045504473,0.00016281432,0.00042967775,0.0007878982,0.00017620248],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018745885,0.0019502215,0.0010680539,0.0066455184,0.0014938,0.0036403167,0.0016951722,0.0014085076,0.037200913],"category_scores_gemma":[0.007872437,0.0013375501,0.0008754915,0.006012694,0.0007313044,0.013284072,0.0048596514,0.002190015,0.05351034],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038589875,0.00010963315,0.0018581658,0.0032268628,0.00013791633,0.0006397309,0.001483224,0.0018608996,0.011038186,0.070496276,0.6297829,0.2789803],"study_design_scores_gemma":[0.000029969175,0.000025405638,0.0007516993,0.0002950037,0.00003743861,0.00034810635,0.0002850393,0.0021875198,0.004871569,0.022608325,0.968511,0.000048882834],"about_ca_topic_score_codex":0.0050150836,"about_ca_topic_score_gemma":0.006344342,"teacher_disagreement_score":0.037200913,"about_ca_system_score_codex":0.0008579882,"about_ca_system_score_gemma":0.0031997978,"threshold_uncertainty_score":0.12444937},"labels":[],"label_agreement":null},{"id":"W2575224319","doi":"","title":"Capturing Pragmatic Knowledge in Article Usage Prediction using LSTMs.","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Interpretability; Computer science; Coreference; Artificial intelligence; Task (project management); Machine learning; Mechanism (biology); Recurrent neural network; Natural language processing; Long short term memory; Artificial neural network; Resolution (logic)","score_opus":0.04258849303752668,"score_gpt":0.33589017408603444,"score_spread":0.29330168104850773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2575224319","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.378333,0.002924536,0.58922976,0.002421591,0.00045761664,0.00023856133,0.0038621568,0.007980743,0.014551967],"genre_scores_gemma":[0.905984,0.0004858828,0.088192746,0.00016567377,0.00012017958,0.000094037576,0.0025892751,0.0001677663,0.0022003925],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992555,0.000290092,0.000056592744,0.00020219614,0.00013182574,0.00006377717],"domain_scores_gemma":[0.9964406,0.0024017717,0.00039957938,0.00021837941,0.00046094053,0.000078805984],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013271462,0.0008343553,0.0003919166,0.0014700742,0.000326522,0.0012509866,0.0010205524,0.0010732159,0.0021836741],"category_scores_gemma":[0.01060308,0.00043155416,0.0004821419,0.0012321074,0.00038165392,0.0034637265,0.0007403807,0.0014256951,0.0011142865],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071927434,0.0004430556,0.025434937,0.0013959596,0.0003934377,0.0009449815,0.0013690677,0.13803978,0.06548867,0.0110616805,0.013448809,0.7412604],"study_design_scores_gemma":[0.000014545089,0.000059000096,0.004229909,0.000055818793,0.000054742446,0.000120305216,0.00017003415,0.97183496,0.008592524,0.011729555,0.0031121972,0.00002638152],"about_ca_topic_score_codex":0.0051003965,"about_ca_topic_score_gemma":0.012786793,"teacher_disagreement_score":0.0051003965,"about_ca_system_score_codex":0.0007990486,"about_ca_system_score_gemma":0.00077859557,"threshold_uncertainty_score":0.010141432},"labels":[],"label_agreement":null},{"id":"W2577720462","doi":"","title":"Named Entity Disambiguation for little known referents: a topic-based approach","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Referent; Exploit; Task (project management); Property (philosophy); Named-entity recognition; Natural language processing; Entity linking; Artificial intelligence; Information retrieval; Training set; Named entity; Linguistics; Knowledge base","score_opus":0.09194997485170264,"score_gpt":0.3341074863322806,"score_spread":0.24215751148057796,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2577720462","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011312237,0.0013296204,0.9804185,0.0007875542,0.00022014102,0.00012867694,0.00072005665,0.0032065401,0.0018767164],"genre_scores_gemma":[0.24928297,0.0016733521,0.7327743,0.00069201266,0.001032024,0.0004247628,0.0061770305,0.0010457304,0.0068977415],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9941949,0.0017809536,0.0005309291,0.0020761576,0.0011027304,0.00031428356],"domain_scores_gemma":[0.98966837,0.0056091757,0.00068763475,0.0020181723,0.0017111066,0.0003054929],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062201777,0.0016567499,0.0024793586,0.009488828,0.002800216,0.0037779182,0.0046927542,0.0029983376,0.0030813725],"category_scores_gemma":[0.014429288,0.0012298749,0.0027074856,0.008121777,0.0012881155,0.009456827,0.0049225083,0.003011609,0.0044672983],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062236085,0.00066686433,0.0112306615,0.0011153544,0.00065530796,0.0013828077,0.004280854,0.03827206,0.030209402,0.0416601,0.033995934,0.8359083],"study_design_scores_gemma":[0.000114797476,0.00019709943,0.00552349,0.00029066138,0.0006398302,0.001979173,0.0020648604,0.7761338,0.032353573,0.096247174,0.0842006,0.00025507508],"about_ca_topic_score_codex":0.0032183377,"about_ca_topic_score_gemma":0.006890372,"teacher_disagreement_score":0.009488828,"about_ca_system_score_codex":0.00091281027,"about_ca_system_score_gemma":0.002376503,"threshold_uncertainty_score":0.032895803},"labels":[],"label_agreement":null},{"id":"W2579534470","doi":"","title":"Training Data Enrichment for Infrequent Discourse Relations","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Parsing; Computer science; Training set; Relation (database); Natural language processing; Artificial intelligence; Training (meteorology); Confidence interval; Quality (philosophy); Machine learning; Data mining; Statistics","score_opus":0.1427233374709916,"score_gpt":0.40637640218257554,"score_spread":0.2636530647115839,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2579534470","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33163306,0.0012241778,0.6375189,0.0015459607,0.0003439669,0.0007044203,0.0025693637,0.019973714,0.004486403],"genre_scores_gemma":[0.47832695,0.00019304316,0.5049353,0.00064299227,0.00014967001,0.00072177214,0.010426529,0.0010708482,0.0035328374],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99207824,0.0034423806,0.0006091158,0.0025194378,0.00094415585,0.00040674573],"domain_scores_gemma":[0.94682914,0.038767457,0.0017767797,0.007707406,0.004270123,0.0006490824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009819563,0.0019215707,0.0016492169,0.002508579,0.0017553864,0.0016713565,0.0026755268,0.003472533,0.0031077168],"category_scores_gemma":[0.043632623,0.0010359377,0.001568262,0.0021134927,0.0013829153,0.004070283,0.0035791784,0.0041537695,0.0032104936],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015062721,0.0023785082,0.049404297,0.0010738946,0.0002886457,0.0013622941,0.0029370964,0.053795483,0.067340426,0.006139383,0.027295647,0.78647804],"study_design_scores_gemma":[0.00015824207,0.00059235527,0.009649676,0.00023885227,0.00023224369,0.0008444557,0.001198024,0.8404607,0.10549368,0.0147949355,0.026238633,0.00009831866],"about_ca_topic_score_codex":0.002563991,"about_ca_topic_score_gemma":0.0054675345,"teacher_disagreement_score":0.009819563,"about_ca_system_score_codex":0.00094772334,"about_ca_system_score_gemma":0.0021765863,"threshold_uncertainty_score":0.05193144},"labels":[],"label_agreement":null},{"id":"W2579773546","doi":"","title":"Reddit Temporal N-gram Corpus and its Applications on Paraphrase and Semantic Similarity in Social Media using a Topic-based Latent Semantic Analysis.","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Paraphrase; Computer science; Latent semantic analysis; SemEval; Natural language processing; Artificial intelligence; Social media; Semantic similarity; Similarity (geometry); n-gram; Probabilistic latent semantic analysis; Text corpus; Information retrieval; Language model; World Wide Web","score_opus":0.08700156074749933,"score_gpt":0.3337642859727162,"score_spread":0.24676272522521686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2579773546","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44182605,0.004779739,0.40748218,0.0024836445,0.001251,0.0017842584,0.09786966,0.016398208,0.026125243],"genre_scores_gemma":[0.5147017,0.0009896682,0.33904642,0.00026228226,0.0003332823,0.002542671,0.13378134,0.0009644694,0.0073781135],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9981389,0.00089813635,0.00015492948,0.00030378852,0.00042507253,0.000079271784],"domain_scores_gemma":[0.9943094,0.0029715463,0.00032998182,0.0010660248,0.0011261891,0.00019683663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014649419,0.000792904,0.00053502957,0.004571416,0.0017057266,0.0010965274,0.0011068844,0.00096268795,0.005859961],"category_scores_gemma":[0.012498006,0.00028789026,0.0005627694,0.004906395,0.00072823954,0.002960093,0.002502844,0.001221627,0.003157371],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021067155,0.0013249174,0.01241132,0.0031779355,0.00025079542,0.002362316,0.004167434,0.016675029,0.061229795,0.0300867,0.14024632,0.7259607],"study_design_scores_gemma":[0.00038205352,0.0007674061,0.044630487,0.00048548458,0.00016786908,0.0033904633,0.0043852073,0.6219219,0.06434038,0.042790443,0.21640138,0.00033693798],"about_ca_topic_score_codex":0.0043808077,"about_ca_topic_score_gemma":0.009902752,"teacher_disagreement_score":0.005859961,"about_ca_system_score_codex":0.0007703181,"about_ca_system_score_gemma":0.0011198667,"threshold_uncertainty_score":0.01960355},"labels":[],"label_agreement":null},{"id":"W2759162401","doi":"","title":"Named Entity Recognition and Hashtag Decomposition to Improve the Classification of Tweets","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec; Université du Québec à Montréal","funders":"","keywords":"WordNet; Computer science; Artificial intelligence; Natural language processing; Preprocessor; Named-entity recognition; Segmentation; Field (mathematics); Task (project management); Information retrieval; Semantics (computer science)","score_opus":0.0829395137147931,"score_gpt":0.3343411440966311,"score_spread":0.25140163038183805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2759162401","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24150814,0.0021721069,0.71530807,0.0013225849,0.0007874202,0.0006183474,0.0054628532,0.024133742,0.008686766],"genre_scores_gemma":[0.52411187,0.00075305905,0.45049724,0.00032342813,0.00030810368,0.00027492674,0.015440213,0.00047468327,0.007816541],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986726,0.00040273336,0.00017349534,0.0003228578,0.00027692132,0.00015151594],"domain_scores_gemma":[0.99658597,0.0013704859,0.0003719011,0.00052330375,0.0010177459,0.00013060105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016448481,0.0010809359,0.0007878874,0.0055133067,0.00074044673,0.0017176085,0.00088476983,0.0010575756,0.0025558916],"category_scores_gemma":[0.004966146,0.00027762252,0.0010684483,0.0043633967,0.00033215908,0.004296219,0.00093681156,0.0010626945,0.0049005128],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001135274,0.0007552558,0.022905488,0.0005453112,0.00033289348,0.00049698714,0.00083550095,0.012818293,0.07755624,0.00754678,0.024395544,0.8506764],"study_design_scores_gemma":[0.00007785489,0.00036622124,0.03187007,0.00009507802,0.00036979557,0.00065509055,0.0010664214,0.7912063,0.1152668,0.017876668,0.040992647,0.0001571258],"about_ca_topic_score_codex":0.0036345446,"about_ca_topic_score_gemma":0.0050314707,"teacher_disagreement_score":0.0055133067,"about_ca_system_score_codex":0.0005720077,"about_ca_system_score_gemma":0.00078473665,"threshold_uncertainty_score":0.008698881},"labels":[],"label_agreement":null},{"id":"W2759598007","doi":"","title":"UQAM-NTL: Named entity recognition in Twitter messages.","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec; Université du Québec à Montréal","funders":"","keywords":"Conditional random field; Named-entity recognition; Computer science; Task (project management); Conjunction (astronomy); Artificial intelligence; Natural language processing; Named entity; Entity linking; Information retrieval; Machine learning; Knowledge base; Engineering","score_opus":0.0937972724176686,"score_gpt":0.3230069221583068,"score_spread":0.22920964974063818,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2759598007","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023659162,0.0009198162,0.14075145,0.0012572564,0.0007275176,0.0012287003,0.25287557,0.5577907,0.02078977],"genre_scores_gemma":[0.13361219,0.00065270445,0.2727251,0.00080338353,0.00031302994,0.0022193932,0.5464757,0.00931306,0.033885468],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998669,0.00032943234,0.00015569862,0.00036536093,0.00036530764,0.00011516685],"domain_scores_gemma":[0.9978136,0.0006607587,0.0002652151,0.0006858767,0.000425902,0.0001485513],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017530058,0.0016057907,0.0010465836,0.0035284194,0.00090121705,0.001548775,0.001795361,0.0014209801,0.02592803],"category_scores_gemma":[0.008590053,0.0004967052,0.00057394203,0.001839298,0.00039004313,0.0048878123,0.0026888265,0.0009231648,0.042956788],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013451176,0.00027904275,0.010294253,0.0015163742,0.000162502,0.00093008974,0.00075454405,0.0040969523,0.031902086,0.004718136,0.6202483,0.32375252],"study_design_scores_gemma":[0.00030129417,0.00041835796,0.013321496,0.00033265154,0.00013224724,0.001031351,0.00070850167,0.28130808,0.1282512,0.01159046,0.5623227,0.00028165133],"about_ca_topic_score_codex":0.0077865543,"about_ca_topic_score_gemma":0.008304489,"teacher_disagreement_score":0.02592803,"about_ca_system_score_codex":0.0009729012,"about_ca_system_score_gemma":0.0011455384,"threshold_uncertainty_score":0.08673787},"labels":[],"label_agreement":null},{"id":"W2785621828","doi":"","title":"Lexfom: a lexical functions ontology model","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Université du Québec","funders":"","keywords":"Syntagmatic analysis; Computer science; Lexical item; Natural language processing; Lexical density; Lexical grammar; Lexical choice; Lexical functional grammar; Artificial intelligence; Relation (database); Function (biology); Ontology; Linguistics; Representation (politics); Perspective (graphical); Generative grammar; Phrase structure rules","score_opus":0.0599011662098063,"score_gpt":0.34416787995760867,"score_spread":0.2842667137478024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2785621828","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008918064,0.0004546549,0.9370622,0.001449426,0.00014755853,0.0004883497,0.014382207,0.017548842,0.019548643],"genre_scores_gemma":[0.12823229,0.0009894954,0.8105661,0.0009022403,0.00012782236,0.0014590529,0.03350372,0.0031998213,0.02101947],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9992539,0.00014803081,0.000117217096,0.00018536262,0.00021961151,0.000075787866],"domain_scores_gemma":[0.9994142,0.00015519239,0.00004883145,0.00017544912,0.00016235864,0.00004393039],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010655649,0.001079211,0.00067672296,0.0033197762,0.0012600374,0.0038797173,0.0025883017,0.0016483237,0.012869513],"category_scores_gemma":[0.003029662,0.00088040996,0.0023644397,0.0021670607,0.00095528405,0.008334832,0.0022469864,0.0020781197,0.005042222],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023571971,0.00026041528,0.0029019113,0.0007968387,0.00014899173,0.00065555546,0.0013348218,0.026373891,0.006025353,0.67964894,0.053366415,0.22825114],"study_design_scores_gemma":[0.000078639205,0.000057818317,0.0009803739,0.00028051002,0.000120520155,0.00057835056,0.00042346134,0.1992463,0.0057584317,0.29769707,0.49468932,0.00008913242],"about_ca_topic_score_codex":0.014732557,"about_ca_topic_score_gemma":0.014305199,"teacher_disagreement_score":0.014732557,"about_ca_system_score_codex":0.0027805679,"about_ca_system_score_gemma":0.0032134494,"threshold_uncertainty_score":0.043052793},"labels":[],"label_agreement":null},{"id":"W2787587581","doi":"","title":"A Proposal for combining “general” and specialized frames","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"FrameNet; Merge (version control); Computer science; Domain (mathematical analysis); Resource (disambiguation); Semantics (computer science); Natural language processing; Representation (politics); Artificial intelligence; Information retrieval; Programming language; Parsing; Mathematics","score_opus":0.034748774668634734,"score_gpt":0.33961479867365135,"score_spread":0.3048660240050166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2787587581","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004866407,0.00026347817,0.96711975,0.0016262844,0.00027633377,0.00015923071,0.000146613,0.00074069184,0.024801165],"genre_scores_gemma":[0.13194166,0.00032835366,0.8545499,0.000908018,0.00039772197,0.00042963578,0.00055257103,0.00055476493,0.01033742],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99415827,0.0022435223,0.00048807912,0.0017728219,0.00077096676,0.0005664863],"domain_scores_gemma":[0.99515027,0.0012392352,0.0003112397,0.0020190964,0.0008269261,0.00045316305],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006954605,0.001591776,0.0013275703,0.004524835,0.003136967,0.006902109,0.0046980553,0.003696007,0.012064948],"category_scores_gemma":[0.009133885,0.0013727759,0.002319356,0.004154444,0.007219665,0.026086394,0.008903231,0.0040552625,0.0029582474],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025436231,0.000014825253,0.00023055947,0.00005283504,0.000014457953,0.000066386056,0.00091964705,0.00049752643,0.0007604071,0.9716713,0.0019860237,0.023760637],"study_design_scores_gemma":[0.00003394626,0.00007348066,0.00036333702,0.00016004835,0.00008745105,0.0004133041,0.0012921329,0.01706614,0.0023929859,0.81186587,0.16618837,0.000062989886],"about_ca_topic_score_codex":0.0048107454,"about_ca_topic_score_gemma":0.0050713345,"teacher_disagreement_score":0.012064948,"about_ca_system_score_codex":0.0029235221,"about_ca_system_score_gemma":0.0035683927,"threshold_uncertainty_score":0.040361226},"labels":[],"label_agreement":null},{"id":"W2814750830","doi":"","title":"Learning Emotion-enriched Word Representations","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Similarity (geometry); Word (group theory); Natural language processing; Computer science; Meaning (existential); Affect (linguistics); Representation (politics); Artificial intelligence; Emotion classification; Contrast (vision); Psychology; Cognitive psychology; Linguistics; Communication","score_opus":0.057740884668665775,"score_gpt":0.35367466296883426,"score_spread":0.2959337783001685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2814750830","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1988266,0.0010424444,0.7910687,0.00051232777,0.00032395968,0.00015372675,0.0014158697,0.0029792937,0.0036770692],"genre_scores_gemma":[0.7955788,0.00076080946,0.19350953,0.00023943171,0.00019207201,0.00027809513,0.0047874725,0.00015257849,0.004501216],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996631,0.000073016614,0.000025431631,0.000149539,0.000045999906,0.000043008047],"domain_scores_gemma":[0.9995515,0.00017043932,0.000056139303,0.0000774562,0.00012329634,0.000021116906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00035591467,0.0010874553,0.0004566773,0.0009804579,0.00020217376,0.00073376513,0.0008016798,0.0008409607,0.0022518751],"category_scores_gemma":[0.0023432192,0.00019130777,0.00068014715,0.0009462062,0.00027738413,0.00197978,0.000851773,0.0011106118,0.001029467],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049729645,0.00040992268,0.003740117,0.00034051618,0.00018450874,0.0002501267,0.0004701553,0.056836113,0.052562557,0.01055502,0.013619407,0.8605343],"study_design_scores_gemma":[0.000058471094,0.00020425396,0.002228771,0.000045660316,0.00012294974,0.00014265101,0.00022507312,0.94537604,0.014279632,0.03312405,0.004159208,0.00003319436],"about_ca_topic_score_codex":0.0007891998,"about_ca_topic_score_gemma":0.0012719877,"teacher_disagreement_score":0.0022518751,"about_ca_system_score_codex":0.00040506065,"about_ca_system_score_gemma":0.00032870428,"threshold_uncertainty_score":0.0075332522},"labels":[],"label_agreement":null},{"id":"W2816262648","doi":"","title":"Farewell Freebase: Migrating the SimpleQuestions Dataset to DBpedia.","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Leverage (statistics); Question answering; Knowledge graph; Information retrieval; Benchmark (surveying); Entity linking; Task (project management); Simple (philosophy); Graph; World Wide Web; Knowledge base; Theoretical computer science; Artificial intelligence","score_opus":0.07890198444019045,"score_gpt":0.36377097578953593,"score_spread":0.2848689913493455,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2816262648","genre_codex":"dataset","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017401932,0.0016951718,0.009498534,0.0011757609,0.00047769304,0.0004464388,0.93909186,0.017922837,0.012289765],"genre_scores_gemma":[0.014149074,0.00028060342,0.01578847,0.00035260763,0.000034194796,0.00027387333,0.9673031,0.00057177624,0.0012462791],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9965287,0.00076141173,0.00043797898,0.0011170573,0.00093761674,0.00021724292],"domain_scores_gemma":[0.99260265,0.0023566731,0.00050498825,0.0021675685,0.001708113,0.0006599839],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024799365,0.0020080474,0.0009878329,0.008066264,0.0023137277,0.003215827,0.0036791137,0.0026229224,0.009437288],"category_scores_gemma":[0.01734786,0.0007483185,0.0014942375,0.007305928,0.0009610266,0.005626526,0.003969439,0.0029743367,0.008098693],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034235689,0.00038843465,0.0051211677,0.0031214429,0.00028425342,0.000446297,0.0007706146,0.0039012586,0.0029440888,0.007295785,0.9403528,0.035031643],"study_design_scores_gemma":[0.0003500033,0.000118920114,0.012370725,0.00047614003,0.00012471685,0.00056099176,0.0013848193,0.021178383,0.007407871,0.014216414,0.9416279,0.00018310717],"about_ca_topic_score_codex":0.069076136,"about_ca_topic_score_gemma":0.09977357,"teacher_disagreement_score":0.069076136,"about_ca_system_score_codex":0.002605333,"about_ca_system_score_gemma":0.0032135013,"threshold_uncertainty_score":0.13734812},"labels":[],"label_agreement":null},{"id":"W2847160827","doi":"","title":"Reproducing and Regularizing the SCRN Model","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Dropout (neural networks); Tying; Computer science; Language model; Task (project management); Artificial intelligence; Data modeling; Machine learning; Algorithm; Engineering","score_opus":0.07632542245516838,"score_gpt":0.3238737291177107,"score_spread":0.2475483066625423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2847160827","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.045707133,0.00011119291,0.9482759,0.00031986428,0.0001293001,0.000044462508,0.00032015878,0.001803589,0.003288454],"genre_scores_gemma":[0.74690574,0.00019879869,0.23963735,0.00030223955,0.00010086029,0.000190713,0.0014172589,0.0007113001,0.010535665],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995832,0.000087578934,0.000017431232,0.00016138722,0.0000979853,0.000052507206],"domain_scores_gemma":[0.9993351,0.00019739204,0.000061241524,0.00023535547,0.0001356305,0.00003521752],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008163284,0.00065150874,0.00048780505,0.00031704782,0.0002457387,0.00045949512,0.0014371332,0.00079189474,0.0028185009],"category_scores_gemma":[0.0035232746,0.00033077117,0.00069771917,0.00033341543,0.0005455735,0.0012201883,0.0010311921,0.0014380923,0.0014877109],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011165804,0.00009218569,0.0014485373,0.00010195495,0.000071367904,0.00028808953,0.00016034198,0.82941943,0.0340237,0.044679325,0.004983946,0.08461945],"study_design_scores_gemma":[0.0000037091102,0.000014216019,0.00009772358,0.000002505546,0.00000576455,0.000027926328,0.000004114226,0.9925249,0.0017200515,0.0050422954,0.00055130036,0.0000054722977],"about_ca_topic_score_codex":0.0057529435,"about_ca_topic_score_gemma":0.0082282415,"teacher_disagreement_score":0.0057529435,"about_ca_system_score_codex":0.0004698087,"about_ca_system_score_gemma":0.0010029012,"threshold_uncertainty_score":0.011438906},"labels":[],"label_agreement":null},{"id":"W2848618769","doi":"","title":"Authorship Identification for Literary Book Recommendations","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Authorship Attribution and Profiling","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Reading (process); Recommender system; Pleasure; Computer science; Identification (biology); Style (visual arts); Factor (programming language); Writing style; Qualitative analysis; Information retrieval; World Wide Web; Natural language processing; Artificial intelligence; Qualitative research; Psychology; Linguistics; Literature; Art; Sociology","score_opus":0.11276954657335071,"score_gpt":0.39432093701642357,"score_spread":0.28155139044307287,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2848618769","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24852443,0.0070220935,0.6640635,0.0034457466,0.0014669421,0.0011230335,0.012542356,0.031553164,0.03025878],"genre_scores_gemma":[0.69592845,0.0010777458,0.2756044,0.00029210869,0.00045183479,0.0002178476,0.008792351,0.0003737731,0.017261589],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978028,0.00051571836,0.00020759452,0.00074178335,0.00056051055,0.00017165375],"domain_scores_gemma":[0.9917223,0.0033830996,0.0007185491,0.0016966163,0.0020460014,0.000433442],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002096687,0.00096812996,0.0009009565,0.0057971207,0.0014338086,0.0021347883,0.0013050844,0.0015248052,0.006038892],"category_scores_gemma":[0.015783997,0.00052353414,0.0009393167,0.0031727054,0.00029980292,0.0035105424,0.0011217566,0.0014014969,0.0071836365],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045316745,0.00072953675,0.060161084,0.0005872764,0.00023798297,0.00032463166,0.0009987069,0.010957928,0.007996833,0.0032379576,0.038183697,0.87613124],"study_design_scores_gemma":[0.00009023163,0.00031324633,0.0373199,0.0003924035,0.00034414767,0.0012497363,0.0010993816,0.8548715,0.025238154,0.017777622,0.0611385,0.00016513866],"about_ca_topic_score_codex":0.0052645667,"about_ca_topic_score_gemma":0.01408992,"teacher_disagreement_score":0.006038892,"about_ca_system_score_codex":0.0008642376,"about_ca_system_score_gemma":0.0012890808,"threshold_uncertainty_score":0.02020216},"labels":[],"label_agreement":null},{"id":"W2849292054","doi":"","title":"Automatically Extracting Qualia Relations for the Rich Event Ontology","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Qualia; Computer science; Ontology; Commonsense knowledge; Event (particle physics); Artificial intelligence; Focus (optics); Semantics (computer science); Structuring; Ontology learning; Natural language processing; Upper ontology; Suggested Upper Merged Ontology; Knowledge extraction; Semantic Web; Epistemology; Consciousness","score_opus":0.08858363997313826,"score_gpt":0.3894872429872838,"score_spread":0.30090360301414554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2849292054","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07593526,0.0004658147,0.87947166,0.0016229429,0.00024422753,0.00055378064,0.01537709,0.0064450433,0.019884313],"genre_scores_gemma":[0.27930287,0.0006309386,0.6833987,0.0002688315,0.000075815624,0.00034556034,0.032557737,0.00084585906,0.0025736266],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99872094,0.00020691706,0.00020406397,0.00032877937,0.00043404425,0.000105235034],"domain_scores_gemma":[0.9973648,0.0011323236,0.00036675762,0.0004791164,0.0005501529,0.00010680433],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016493395,0.00067766325,0.00050903205,0.007603145,0.0015567199,0.0025980708,0.0009499292,0.0008025821,0.0045409207],"category_scores_gemma":[0.0078090136,0.00051982066,0.0014935299,0.004382984,0.00080581213,0.0066111768,0.002690648,0.001793984,0.0018418159],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002642147,0.00042414942,0.017590646,0.0014152649,0.0001627865,0.0024512925,0.0052598766,0.011248643,0.04634036,0.429867,0.045578495,0.43939728],"study_design_scores_gemma":[0.00007317478,0.00008068225,0.015653618,0.00056368305,0.00021259068,0.0014231284,0.0054767006,0.25932044,0.032221302,0.3961098,0.28868073,0.0001841838],"about_ca_topic_score_codex":0.007987111,"about_ca_topic_score_gemma":0.018286454,"teacher_disagreement_score":0.007987111,"about_ca_system_score_codex":0.0017932439,"about_ca_system_score_gemma":0.0028189605,"threshold_uncertainty_score":0.01588124},"labels":[],"label_agreement":null},{"id":"W2865911429","doi":"","title":"Abstractive Unsupervised Multi-Document Summarization using Paraphrastic Sentence Fusion","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":53,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Automatic summarization; Computer science; Sentence; Natural language processing; Artificial intelligence; Set (abstract data type); Machine translation; Word embedding; Multi-document summarization; Word (group theory); Information retrieval; Embedding; Linguistics","score_opus":0.09255504555690312,"score_gpt":0.35067864217008043,"score_spread":0.2581235966131773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2865911429","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018355064,0.0009101945,0.9712451,0.00022128792,0.00011178907,0.00016683199,0.0006225855,0.0072473385,0.0011198493],"genre_scores_gemma":[0.1897727,0.00072458456,0.7970082,0.00034314755,0.000349434,0.0003072501,0.0060743783,0.0005512945,0.0048690033],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988194,0.00032943042,0.0001243749,0.0003543253,0.00031309293,0.00005918457],"domain_scores_gemma":[0.9974045,0.0008127111,0.00038971123,0.00051168137,0.0008006206,0.00008063941],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012847204,0.0014694282,0.0013671773,0.0019833946,0.00045852744,0.001104998,0.0013231006,0.0009671329,0.0022918382],"category_scores_gemma":[0.0039166613,0.00035531464,0.0012414479,0.0016408789,0.0003558526,0.0023758374,0.0010276346,0.0012959691,0.002481435],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003737762,0.00033032396,0.0010948894,0.00082284975,0.00029266952,0.00033651828,0.00051307864,0.028880255,0.15129773,0.004211293,0.012095587,0.799751],"study_design_scores_gemma":[0.00010101093,0.0009214451,0.0035138917,0.000079606085,0.00044681405,0.0007933794,0.00033161754,0.7692926,0.18620044,0.012316371,0.02587905,0.00012371186],"about_ca_topic_score_codex":0.0010600868,"about_ca_topic_score_gemma":0.0018231857,"teacher_disagreement_score":0.0022918382,"about_ca_system_score_codex":0.00045356475,"about_ca_system_score_gemma":0.0006494264,"threshold_uncertainty_score":0.0076669455},"labels":[],"label_agreement":null},{"id":"W2875408189","doi":"","title":"The APVA-TURBO Approach To Question Answering in Knowledge Base","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Question answering; Computer science; Correctness; Bottleneck; Knowledge base; Object (grammar); Artificial intelligence; Base (topology); Subject (documents); Information retrieval; Theoretical computer science; Machine learning; Programming language; World Wide Web","score_opus":0.06531592434647242,"score_gpt":0.3406081913215932,"score_spread":0.27529226697512077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2875408189","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004315417,0.00025982465,0.9919853,0.00037404205,0.000028031613,0.000058628513,0.000080165024,0.0019354278,0.0009630673],"genre_scores_gemma":[0.24066839,0.00043367874,0.75274307,0.00058207277,0.00016991462,0.00035281258,0.00080476876,0.00046322547,0.0037820507],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99528235,0.002525405,0.00019102154,0.00091667,0.00082866143,0.00025589118],"domain_scores_gemma":[0.9850085,0.009581022,0.0002882598,0.0034634268,0.001357321,0.00030147578],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061624935,0.0010260835,0.0013692237,0.002046456,0.001024182,0.002548789,0.0054737795,0.0027373547,0.0044509736],"category_scores_gemma":[0.022215897,0.0011806963,0.0018609248,0.0020292578,0.0023854258,0.007513308,0.004650464,0.004600807,0.001947391],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004778168,0.00027462177,0.002046547,0.00065059104,0.00023707867,0.00025396133,0.001082996,0.21938737,0.007949492,0.10112928,0.0116391685,0.65487105],"study_design_scores_gemma":[0.000011976727,0.00004874466,0.00017048721,0.000020942216,0.00002477346,0.00008155451,0.000050141414,0.915821,0.0021593964,0.07884306,0.0027545409,0.000013465886],"about_ca_topic_score_codex":0.0065900967,"about_ca_topic_score_gemma":0.007958282,"teacher_disagreement_score":0.0065900967,"about_ca_system_score_codex":0.0014647655,"about_ca_system_score_gemma":0.002045866,"threshold_uncertainty_score":0.032590747},"labels":[],"label_agreement":null},{"id":"W2888880755","doi":"","title":"Do Character-Level Neural Network Language Models Capture Knowledge of Multiword Expression Compositionality?","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Principle of compositionality; Computer science; Artificial intelligence; Natural language processing; Character (mathematics); Treebank; Expression (computer science); Artificial neural network; Programming language; Annotation","score_opus":0.06360377069358782,"score_gpt":0.3540654245744966,"score_spread":0.2904616538809088,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2888880755","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1854207,0.00089776336,0.80238473,0.001625228,0.00018013513,0.00009253507,0.0007118744,0.0011807518,0.0075062644],"genre_scores_gemma":[0.9132646,0.00069783186,0.08027994,0.00037930222,0.00008473181,0.00016711222,0.0009781008,0.00015353429,0.0039949366],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99956506,0.00014967842,0.000022166325,0.00015975896,0.00005730129,0.00004601953],"domain_scores_gemma":[0.99767035,0.0013340241,0.00028954566,0.0002489868,0.0003803847,0.00007668759],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010711154,0.0007585135,0.0005765493,0.0005699777,0.0002523402,0.0015210385,0.0010954484,0.0008327164,0.0023233604],"category_scores_gemma":[0.007353724,0.00045337435,0.0005262528,0.00052459486,0.0005139881,0.0056121275,0.0006485967,0.0017654097,0.0012580599],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007655436,0.00041580663,0.028925953,0.0007555518,0.00069042674,0.00044156812,0.0010206278,0.34045026,0.06436678,0.0629559,0.004677464,0.4945341],"study_design_scores_gemma":[0.000009184239,0.00004854711,0.002398079,0.000029542914,0.00004591726,0.0000543551,0.00007391087,0.96428317,0.0034299449,0.028309215,0.001295674,0.0000223931],"about_ca_topic_score_codex":0.0028107543,"about_ca_topic_score_gemma":0.0039940323,"teacher_disagreement_score":0.0028107543,"about_ca_system_score_codex":0.0005696893,"about_ca_system_score_gemma":0.00058619154,"threshold_uncertainty_score":0.007772386},"labels":[],"label_agreement":null},{"id":"W2889229100","doi":"","title":"NLP for Conversations: Sentiment, Summarization, and Group Dynamics","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; University of the Fraser Valley","funders":"","keywords":"Automatic summarization; Computer science; Natural language processing; Sentiment analysis; Artificial intelligence; Dynamics (music); Information retrieval; Group (periodic table); Psychology","score_opus":0.029641051308958492,"score_gpt":0.33400512286884554,"score_spread":0.30436407155988704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2889229100","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08452837,0.0016383958,0.8854877,0.004536615,0.0004917681,0.00044633713,0.0077240746,0.0052576396,0.009889054],"genre_scores_gemma":[0.601488,0.0010192469,0.37319246,0.00032710438,0.0009167192,0.0006941078,0.01658889,0.0007849846,0.0049884454],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99582857,0.0023078432,0.00033182662,0.00069848186,0.0006552272,0.00017807771],"domain_scores_gemma":[0.9825434,0.012961539,0.0009928436,0.0016012174,0.0015892233,0.0003117538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004400794,0.00094884593,0.0008258212,0.0025368144,0.0014940177,0.0025575005,0.0010702127,0.0010789571,0.0051638247],"category_scores_gemma":[0.029656868,0.00044421072,0.0008051083,0.0028670838,0.0005975799,0.006166846,0.0020517635,0.0020275754,0.003252694],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009171231,0.00042487917,0.007998055,0.0013369176,0.00024514107,0.00037642338,0.0062032402,0.016714819,0.026298074,0.03730067,0.045698114,0.8564866],"study_design_scores_gemma":[0.00008897978,0.00024361901,0.009763062,0.00025383904,0.00018429795,0.00029153138,0.0037598235,0.74730897,0.016254889,0.17883307,0.042912673,0.000105294406],"about_ca_topic_score_codex":0.0022860994,"about_ca_topic_score_gemma":0.0023020555,"teacher_disagreement_score":0.0051638247,"about_ca_system_score_codex":0.00090152066,"about_ca_system_score_gemma":0.0010714118,"threshold_uncertainty_score":0.023273885},"labels":[],"label_agreement":null},{"id":"W2914220664","doi":"","title":"Deep Models for Arabic Dialect Identification on Benchmarked Data","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Benchmark (surveying); Computer science; Artificial intelligence; Task (project management); Deep learning; Arabic; Natural language processing; Binary classification; Deep neural networks; Identification (biology); Machine learning; Artificial neural network; Recurrent neural network; Test data; Binary number; Speech recognition; Support vector machine; Linguistics; Mathematics; Geography","score_opus":0.10115388688345246,"score_gpt":0.3765717506027052,"score_spread":0.27541786371925275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914220664","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7793805,0.011807667,0.0618514,0.0033831275,0.0021883887,0.0006476797,0.09620999,0.024425967,0.020105287],"genre_scores_gemma":[0.67011416,0.0013390309,0.08052433,0.0009237159,0.00031667895,0.00049372576,0.23421347,0.0006441923,0.011430647],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975515,0.000904968,0.00020474827,0.0006413599,0.00044139335,0.00025599077],"domain_scores_gemma":[0.99372697,0.0025973618,0.00028610558,0.001557822,0.0015385837,0.00029313684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038226657,0.0031538743,0.0011546368,0.0028566427,0.0011630057,0.0017738052,0.0028360605,0.0022790865,0.006617148],"category_scores_gemma":[0.013735471,0.00051750697,0.0013434809,0.0026688003,0.0010683541,0.0030461634,0.00257073,0.003556804,0.005859227],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027233937,0.0025742683,0.019387111,0.0017075799,0.00081503636,0.0008286241,0.0006264319,0.23742059,0.011102963,0.0036541803,0.23026352,0.48889625],"study_design_scores_gemma":[0.00040961913,0.00054991373,0.011789776,0.00023360949,0.00016039368,0.00032465387,0.0008291264,0.93164015,0.016729506,0.008715757,0.028488608,0.00012889814],"about_ca_topic_score_codex":0.031149788,"about_ca_topic_score_gemma":0.040514827,"teacher_disagreement_score":0.031149788,"about_ca_system_score_codex":0.0022522502,"about_ca_system_score_gemma":0.0014017673,"threshold_uncertainty_score":0.061936915},"labels":[],"label_agreement":null},{"id":"W2914752849","doi":"","title":"Cyberbullying Intervention Based on Convolutional Neural Networks","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Bullying, Victimization, and Aggression","field":"Psychology","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Flagging; Computer science; Convolutional neural network; Intervention (counseling); Interface (matter); Process (computing); User interface; Mechanism (biology); Service (business); Human–computer interaction; Natural (archaeology); World Wide Web; Artificial intelligence; Multimedia; Psychology","score_opus":0.048301103511600885,"score_gpt":0.3533375149440351,"score_spread":0.3050364114324342,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914752849","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17724176,0.0014160374,0.79327023,0.0011677982,0.00033668947,0.00020052813,0.0006000622,0.011973276,0.013793668],"genre_scores_gemma":[0.8977245,0.00043219503,0.09080887,0.00029707616,0.000055778724,0.00014844115,0.0005824034,0.00016711363,0.009783629],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998136,0.00003266632,0.000009453841,0.000059278347,0.00004221333,0.000042808104],"domain_scores_gemma":[0.99961805,0.00020376767,0.000042503936,0.000028572093,0.00008352319,0.000023535913],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00040572113,0.0006690941,0.00040879785,0.0005094378,0.00029386018,0.0005837632,0.0010281904,0.0006013196,0.0031605752],"category_scores_gemma":[0.0014370051,0.00026841348,0.00041457193,0.00034006004,0.00027342353,0.0007188174,0.00050533947,0.0008889324,0.0006595096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044514457,0.00054545223,0.0072461423,0.00019880167,0.00017011716,0.0003828306,0.00027387877,0.45035824,0.0263936,0.009567275,0.009507785,0.49491066],"study_design_scores_gemma":[0.0000043389728,0.000026706395,0.0005638315,0.000008981612,0.000014146509,0.000016685679,0.000009667897,0.9941765,0.002828667,0.0017790532,0.0005653747,0.000006035907],"about_ca_topic_score_codex":0.018729163,"about_ca_topic_score_gemma":0.021320904,"teacher_disagreement_score":0.018729163,"about_ca_system_score_codex":0.0011092817,"about_ca_system_score_gemma":0.0007186216,"threshold_uncertainty_score":0.037240267},"labels":[],"label_agreement":null},{"id":"W2962696263","doi":"","title":"Extracting Parallel Sentences with Bidirectional Recurrent Neural Networks to Improve Machine Translation","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Sentence; Artificial intelligence; Task (project management); Natural language processing; Translation (biology); Feature engineering; Recurrent neural network; Parallel corpora; Baseline (sea); Artificial neural network; Feature (linguistics); Feature extraction; Speech recognition; Deep learning","score_opus":0.03912998338648059,"score_gpt":0.3326886547462771,"score_spread":0.2935586713597965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962696263","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09780084,0.001507756,0.8837446,0.00057620637,0.00043487025,0.00021876744,0.00092305016,0.009709714,0.0050841286],"genre_scores_gemma":[0.3819616,0.0008031826,0.6018493,0.00047564175,0.00034040058,0.0003577566,0.0060969265,0.00092613854,0.007189017],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992028,0.00025052196,0.0000871623,0.00019941757,0.00018923,0.00007087732],"domain_scores_gemma":[0.998403,0.0005940583,0.00016211273,0.00024098033,0.00055700826,0.000042816253],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010589669,0.0016512768,0.0010800065,0.0012739786,0.00067139394,0.0008562485,0.0009095888,0.0008769784,0.0033750131],"category_scores_gemma":[0.004423319,0.0005207679,0.0010096051,0.0016446022,0.00037775672,0.0023014315,0.0010618935,0.0012875298,0.0031914392],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057037285,0.00047930502,0.002073808,0.0007271477,0.0003263101,0.0011002667,0.0005443455,0.06104802,0.14057729,0.007311102,0.019637784,0.76560426],"study_design_scores_gemma":[0.00007344663,0.00033904292,0.0013617589,0.00004061663,0.00025553795,0.00036327512,0.0001688803,0.90057826,0.07185357,0.0139908455,0.010914855,0.000059925213],"about_ca_topic_score_codex":0.0024753662,"about_ca_topic_score_gemma":0.0060492824,"teacher_disagreement_score":0.0033750131,"about_ca_system_score_codex":0.00040877933,"about_ca_system_score_gemma":0.0010915883,"threshold_uncertainty_score":0.01129055},"labels":[],"label_agreement":null}]}