{"meta":{"query_hash":"5250998b0ea9","filters":{"venue":"Meeting of the Association for Computational Linguistics"},"cohort_total":22,"direct_labels_cover":0,"predictions_cover":22,"exported":22,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/5250998b0ea9","api":"https://metacan.xera.ac/api/v1/cohort?venue=Meeting+of+the+Association+for+Computational+Linguistics"},"results":[{"id":"W1458921027","doi":"","title":"Ontology-Based Extraction and Summarization of Protein Mutation Impact Information","year":2010,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Automatic summarization; Computer science; Ontology; Information retrieval; Information extraction; Mutation; Pace; Data mining; Geography; Biology; Genetics","score_opus":0.006578853222055952,"score_gpt":0.27714733627980903,"score_spread":0.2705684830577531,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1458921027","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.124486424,0.002692535,0.7964002,0.002377236,0.00044869658,0.0012181416,0.051686566,0.011284981,0.009405237],"genre_scores_gemma":[0.19235627,0.002622599,0.7355883,0.00024192031,0.00025578734,0.0005384114,0.06499907,0.00063277315,0.0027648122],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982942,0.00021968385,0.0004274254,0.00030426524,0.00065981847,0.000094699906],"domain_scores_gemma":[0.9960394,0.0013947418,0.00074695674,0.00048376905,0.001206969,0.00012816038],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017948544,0.0009014318,0.0010169661,0.015455084,0.00076970836,0.0018885075,0.00089680386,0.0006517634,0.0014422587],"category_scores_gemma":[0.008268399,0.0003222896,0.001285783,0.009566513,0.00035871167,0.0030411382,0.0012635654,0.0010051085,0.0008836957],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040458073,0.00040109054,0.01317062,0.0024059454,0.00048700953,0.0014898172,0.0024334576,0.009942805,0.082320556,0.016698316,0.033124886,0.83712095],"study_design_scores_gemma":[0.00025622454,0.00056880387,0.06573948,0.001126444,0.002958806,0.0033804716,0.0052224738,0.2797298,0.17097068,0.12304273,0.34647223,0.0005319024],"about_ca_topic_score_codex":0.0042071613,"about_ca_topic_score_gemma":0.0053341785,"teacher_disagreement_score":0.015455084,"about_ca_system_score_codex":0.0009955373,"about_ca_system_score_gemma":0.0023982134,"threshold_uncertainty_score":0.0094922185},"labels":[],"label_agreement":null},{"id":"W172656279","doi":"","title":"RALI: Automatic Weighting of Text Window Distances","year":2010,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Window (computing); Weighting; Word (group theory); Computer science; SemEval; Artificial intelligence; Task (project management); Word-sense disambiguation; Natural language processing; Limit (mathematics); Pattern recognition (psychology); Mathematics","score_opus":0.007342829639935821,"score_gpt":0.2629071066691153,"score_spread":0.2555642770291795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W172656279","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026656264,0.0012140092,0.906124,0.00013859603,0.00030894106,0.0004341748,0.0021715965,0.061735228,0.0012171015],"genre_scores_gemma":[0.11563235,0.00035284652,0.8737032,0.00008963068,0.000118555676,0.0005541122,0.003967041,0.0028764994,0.0027058728],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977822,0.0005263804,0.0002233024,0.00092876184,0.00042783935,0.00011160186],"domain_scores_gemma":[0.99552417,0.0020524287,0.0005056604,0.0010254506,0.00072458136,0.00016768056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028533102,0.0022907876,0.0015239163,0.004078322,0.0006956268,0.0015617425,0.0031786105,0.0012259942,0.004682442],"category_scores_gemma":[0.011748752,0.00093595946,0.0011223519,0.0022796893,0.0004621179,0.0043453104,0.002195255,0.001860302,0.0041922173],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010121058,0.00024116637,0.0029005324,0.000771912,0.00028729104,0.00013613158,0.00045166677,0.011573001,0.058807224,0.005057437,0.02420236,0.89455926],"study_design_scores_gemma":[0.0002498229,0.0003893123,0.004641346,0.00007726997,0.000200696,0.00039644772,0.00022493483,0.8376523,0.11169055,0.014052149,0.030250473,0.00017470129],"about_ca_topic_score_codex":0.0017493622,"about_ca_topic_score_gemma":0.0035968428,"teacher_disagreement_score":0.004682442,"about_ca_system_score_codex":0.00069770153,"about_ca_system_score_gemma":0.0008316221,"threshold_uncertainty_score":0.015664339},"labels":[],"label_agreement":null},{"id":"W2096565906","doi":"","title":"Alignment-Based Discriminative String Similarity","year":2007,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Discriminative model; Coreference; Artificial intelligence; Computer science; Substring; String metric; Similarity (geometry); Character (mathematics); Natural language processing; Longest common subsequence problem; String (physics); Word (group theory); Pattern recognition (psychology); Transliteration; Heuristic; String searching algorithm; Mathematics; Pattern matching; Resolution (logic); Algorithm; Set (abstract data type)","score_opus":0.016287383324153314,"score_gpt":0.2953723765151392,"score_spread":0.2790849931909859,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096565906","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.071590476,0.0004947179,0.92107713,0.00008972681,0.00006303943,0.00014085282,0.00049438543,0.0024366227,0.0036129039],"genre_scores_gemma":[0.65871793,0.00021564485,0.33593118,0.00013686359,0.000090831214,0.00016617337,0.0022112723,0.00026294275,0.002267076],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9973628,0.0006219286,0.00016836106,0.00069978915,0.0009777171,0.00016941133],"domain_scores_gemma":[0.9960316,0.0011260952,0.00054293894,0.0012828233,0.000843252,0.00017325483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015423504,0.0006003175,0.0015307086,0.0038469615,0.00068933656,0.0011024094,0.0015710277,0.0007734903,0.0025075197],"category_scores_gemma":[0.007312349,0.0002540779,0.0005424664,0.004968217,0.00085128366,0.002222912,0.0016423991,0.000862533,0.0017948373],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004615589,0.00038852202,0.013246618,0.00041860636,0.00015520724,0.00018255835,0.00024567105,0.034781102,0.08358547,0.031604696,0.007315861,0.8276141],"study_design_scores_gemma":[0.00007156301,0.0008021127,0.01997499,0.00003487004,0.00010344212,0.0017160926,0.00019462398,0.8387704,0.073135614,0.052172024,0.012901578,0.0001227065],"about_ca_topic_score_codex":0.0008711013,"about_ca_topic_score_gemma":0.0015194186,"teacher_disagreement_score":0.0038469615,"about_ca_system_score_codex":0.00055124256,"about_ca_system_score_gemma":0.000798322,"threshold_uncertainty_score":0.00838846},"labels":[],"label_agreement":null},{"id":"W2096960653","doi":"","title":"Selecting Query Term Alternations for Web Search by Exploiting Query Contexts","year":2008,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Query expansion; Computer science; Bigram; Web query classification; Web search query; Information retrieval; Query optimization; Query language; Sargable; Selection (genetic algorithm); Term (time); RDF query language; Context (archaeology); Search engine; Word (group theory); Natural language processing; Artificial intelligence","score_opus":0.02470986278797762,"score_gpt":0.2848725438985707,"score_spread":0.2601626811105931,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096960653","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.65075463,0.005428444,0.3318554,0.0006286401,0.0001276106,0.0008941311,0.0005953457,0.00466634,0.005049529],"genre_scores_gemma":[0.7845204,0.0010299041,0.21045333,0.0002108848,0.00028137115,0.00032593464,0.00110019,0.00040192207,0.0016760846],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988952,0.0003553624,0.00013316765,0.00023389778,0.00028464638,0.00009778423],"domain_scores_gemma":[0.99722433,0.0017118352,0.00022528281,0.00028421957,0.00041921847,0.0001351521],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012514467,0.0010649003,0.0014510688,0.0038920785,0.00070718996,0.0009618033,0.0008228598,0.0006048866,0.0016443634],"category_scores_gemma":[0.0067106364,0.00044235383,0.0005431543,0.0028083299,0.0005072027,0.0021296619,0.0010706787,0.0007898942,0.001021419],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020878445,0.0005831356,0.011521622,0.0005008046,0.00012284267,0.00048214436,0.0005999417,0.011101766,0.21683824,0.0027015419,0.004274617,0.7491856],"study_design_scores_gemma":[0.0007564157,0.002103671,0.030383635,0.00013079886,0.0010560667,0.0029879855,0.0010860366,0.7874183,0.13960384,0.013562615,0.020594187,0.00031636198],"about_ca_topic_score_codex":0.0022346876,"about_ca_topic_score_gemma":0.00660232,"teacher_disagreement_score":0.0038920785,"about_ca_system_score_codex":0.00039476523,"about_ca_system_score_gemma":0.0010271261,"threshold_uncertainty_score":0.006618321},"labels":[],"label_agreement":null},{"id":"W2097120204","doi":"","title":"Towards Robust Abstractive Multi-Document Summarization: A Caseframe Analysis of Centrality and Domain","year":2013,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Automatic summarization; Centrality; Computer science; Redundancy (engineering); Sentence; Natural language processing; Information retrieval; Domain (mathematical analysis); Artificial intelligence; Abstraction; Multi-document summarization","score_opus":0.020833600090225578,"score_gpt":0.26804744871907904,"score_spread":0.24721384862885346,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2097120204","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040767953,0.0007591442,0.9560688,0.00045788867,0.000024643581,0.000094062605,0.00013764395,0.0004161739,0.0012736627],"genre_scores_gemma":[0.4565983,0.00054351887,0.5402343,0.000089558474,0.00015024506,0.00014029002,0.0007705925,0.00021748456,0.0012556737],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997279,0.0013106664,0.00018278605,0.0005679709,0.00054011546,0.00011949417],"domain_scores_gemma":[0.98500454,0.008888086,0.001572664,0.0018312485,0.0024471579,0.00025624907],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035965866,0.00079754536,0.0007488864,0.0049431003,0.000990977,0.0025456802,0.0011394157,0.00109075,0.0012047368],"category_scores_gemma":[0.02187389,0.00039624583,0.0008011211,0.0032397474,0.0012543853,0.004688827,0.0015727977,0.0012358986,0.0004541005],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007882637,0.0002837,0.0116812,0.00092258287,0.00028850694,0.00087327766,0.0049879327,0.11355018,0.04971266,0.12081436,0.0065386295,0.6895588],"study_design_scores_gemma":[0.00004944402,0.00021264565,0.005276733,0.00008057498,0.00017902837,0.0003835017,0.001204182,0.84313446,0.023100654,0.11394077,0.0123599395,0.00007798223],"about_ca_topic_score_codex":0.0024256583,"about_ca_topic_score_gemma":0.0024922465,"teacher_disagreement_score":0.0049431003,"about_ca_system_score_codex":0.0009827525,"about_ca_system_score_gemma":0.00085353845,"threshold_uncertainty_score":0.019020796},"labels":[],"label_agreement":null},{"id":"W2099233844","doi":"","title":"Even the Abstract have Color: Consensus in Word-Colour Associations","year":2011,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Categorization, perception, and language","field":"Psychology","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Lexicon; Computer science; Word Association; Focus (optics); Component (thermodynamics); Crowdsourcing; Word (group theory); Resource (disambiguation); Key (lock); Product (mathematics); Association (psychology); Visualization; Semantics (computer science); Natural language processing; Coherence (philosophical gambling strategy); Quality (philosophy); Artificial intelligence; Linguistics; World Wide Web; Psychology; Mathematics","score_opus":0.045509232035914296,"score_gpt":0.31724088188839533,"score_spread":0.27173164985248105,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099233844","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31829673,0.0023220552,0.6216662,0.0050917724,0.00074474217,0.000388285,0.0013580956,0.0018080837,0.04832406],"genre_scores_gemma":[0.89417064,0.00046554193,0.09967756,0.00059270393,0.0001216429,0.00029935554,0.000988311,0.00054598355,0.0031383592],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98747694,0.0059334976,0.0013276179,0.00283377,0.0020642516,0.0003638856],"domain_scores_gemma":[0.9456494,0.033462226,0.003898442,0.0072626895,0.008807111,0.0009200985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009702478,0.00052848796,0.0006516836,0.0033120324,0.0029474306,0.0053466684,0.0009428904,0.0014186801,0.0049562873],"category_scores_gemma":[0.06898937,0.00062892697,0.0007060119,0.0036694778,0.0048128003,0.009835933,0.00541306,0.002061763,0.0015438315],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013126791,0.00018972237,0.034609493,0.0028530015,0.00033321153,0.0012972283,0.091498315,0.0053790133,0.06015683,0.3548822,0.029962184,0.41752607],"study_design_scores_gemma":[0.00013981972,0.00016552556,0.043980796,0.0005552818,0.00035691692,0.0010991419,0.023925224,0.026948312,0.017397424,0.73004943,0.1550304,0.00035178376],"about_ca_topic_score_codex":0.0025799884,"about_ca_topic_score_gemma":0.0027392947,"teacher_disagreement_score":0.009702478,"about_ca_system_score_codex":0.0013794196,"about_ca_system_score_gemma":0.001413768,"threshold_uncertainty_score":0.05131221},"labels":[],"label_agreement":null},{"id":"W2099242427","doi":"","title":"Entity-Based Local Coherence Modelling Using Topological Fields","year":2010,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Coherence (philosophical gambling strategy); Computer science; Sentence; Component (thermodynamics); Topology (electrical circuits); Grid; Natural language; Natural language processing; Field (mathematics); Artificial intelligence; Language model; Theoretical computer science; Mathematics; Physics; Pure mathematics","score_opus":0.022983696173067523,"score_gpt":0.29087224003836043,"score_spread":0.2678885438652929,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099242427","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05698764,0.0001418149,0.9385676,0.0002056856,0.000022116114,0.00009298949,0.00037537076,0.0016642377,0.001942639],"genre_scores_gemma":[0.7269127,0.00013371906,0.27028677,0.00004423334,0.000019514942,0.00013022593,0.00083434774,0.00021498858,0.0014235975],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920887,0.00037340782,0.00005538167,0.00016704004,0.00015136918,0.000043908465],"domain_scores_gemma":[0.9961398,0.0026373726,0.00040349178,0.00040049403,0.000333342,0.00008550796],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017639977,0.0004992763,0.0004229429,0.0015549066,0.00051336654,0.0014534115,0.0009400075,0.00063291605,0.002955499],"category_scores_gemma":[0.006124229,0.00039937923,0.00082030916,0.001048249,0.0007518596,0.00425284,0.0010516379,0.00075954566,0.00046960925],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046246158,0.00014950981,0.00595338,0.00033328493,0.00009735518,0.00044769954,0.0012973804,0.61812544,0.012993755,0.23496929,0.0028459348,0.1223245],"study_design_scores_gemma":[0.000030966305,0.00005028101,0.00045027208,0.000012149997,0.000024899786,0.00004371049,0.0000745316,0.95421165,0.0035443283,0.038835883,0.0027046474,0.000016789829],"about_ca_topic_score_codex":0.0041893753,"about_ca_topic_score_gemma":0.0061545493,"teacher_disagreement_score":0.0041893753,"about_ca_system_score_codex":0.00084597623,"about_ca_system_score_gemma":0.0007361128,"threshold_uncertainty_score":0.009887099},"labels":[],"label_agreement":null},{"id":"W2110750141","doi":"","title":"Automatic detection of deception in child-produced speech using syntactic complexity features","year":2013,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Deception detection and forensic psychology","field":"Psychology","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Utterance; Deception; Sentence; Support vector machine; Computer science; Set (abstract data type); Artificial intelligence; Random forest; Measure (data warehouse); Natural language processing; Speech recognition; Pattern recognition (psychology); Machine learning; Psychology; Data mining; Social psychology","score_opus":0.03243081732255438,"score_gpt":0.32348086160658635,"score_spread":0.291050044284032,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2110750141","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97412485,0.00017336525,0.024427284,0.00007203029,0.000011595049,0.000030053621,0.00019941683,0.00015927502,0.0008020561],"genre_scores_gemma":[0.9885106,0.00008027356,0.010866846,0.00000883317,0.000009491314,0.000018092229,0.00032669707,0.0000128060865,0.00016631513],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99876213,0.0005936299,0.00009358568,0.00014699748,0.0002868549,0.00011682724],"domain_scores_gemma":[0.9901175,0.0067432625,0.0014652179,0.0004677344,0.0010335079,0.00017283307],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022540477,0.0004236963,0.0003970269,0.0014461488,0.00030736014,0.0009201335,0.00042169268,0.0005523149,0.00086289796],"category_scores_gemma":[0.011013056,0.00016990237,0.0002695578,0.00044816028,0.0003571478,0.0010467925,0.0006996521,0.0005656411,0.0004398132],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015979826,0.00034242932,0.30922037,0.00045308995,0.00023035769,0.0010070596,0.0032433218,0.0273016,0.17584547,0.0028161868,0.001749022,0.47619316],"study_design_scores_gemma":[0.000033219567,0.00042676376,0.4202918,0.00011052563,0.00010754132,0.0015446723,0.0018664948,0.49068668,0.08049854,0.0028598183,0.0014821092,0.00009188816],"about_ca_topic_score_codex":0.001131197,"about_ca_topic_score_gemma":0.0016172034,"teacher_disagreement_score":0.0022540477,"about_ca_system_score_codex":0.00035036093,"about_ca_system_score_gemma":0.00036492755,"threshold_uncertainty_score":0.011920691},"labels":[],"label_agreement":null},{"id":"W2115791615","doi":"","title":"Deep Learning for NLP (without Magic)","year":2012,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":132,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Artificial intelligence; Feature engineering; Deep learning; Artificial neural network; Paraphrase; Machine learning; Sentiment analysis; Natural language processing; Feature (linguistics); Focus (optics); MAGIC (telescope); Language model","score_opus":0.022720816901939583,"score_gpt":0.2786095360606215,"score_spread":0.25588871915868194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115791615","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011612065,0.032425556,0.9131408,0.009356223,0.0024057762,0.00009755156,0.0008143042,0.0036204022,0.036978193],"genre_scores_gemma":[0.046705414,0.06686021,0.77927786,0.006987286,0.0039438056,0.00083610957,0.002979167,0.0021894865,0.090220705],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9993895,0.00016211112,0.000047655536,0.00011844681,0.00023200714,0.000050234914],"domain_scores_gemma":[0.99933475,0.0003868128,0.00003873346,0.00008929199,0.00011687104,0.000033512675],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009801683,0.0011946372,0.0005381608,0.00091861363,0.00048439155,0.0027825877,0.001342165,0.0017726759,0.021926563],"category_scores_gemma":[0.0040751896,0.0006503332,0.0008742206,0.0012362924,0.001296812,0.005107338,0.0019384716,0.0042274785,0.010011784],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000039942806,0.000048427737,0.00014878233,0.0010064798,0.00004481854,0.00013761809,0.00017632873,0.014207391,0.0026809787,0.3554519,0.15923527,0.4668221],"study_design_scores_gemma":[0.000016299131,0.000034279143,0.0001663144,0.00047062614,0.000016302813,0.00020670879,0.000045478537,0.052926548,0.0023751718,0.51020676,0.4335012,0.00003428261],"about_ca_topic_score_codex":0.0019265944,"about_ca_topic_score_gemma":0.0021269852,"teacher_disagreement_score":0.021926563,"about_ca_system_score_codex":0.0013482582,"about_ca_system_score_gemma":0.0010135677,"threshold_uncertainty_score":0.07335162},"labels":[],"label_agreement":null},{"id":"W2120158336","doi":"","title":"Cross Lingual Adaptation: An Experiment on Sentiment Classifications","year":2010,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":87,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Task (project management); Artificial intelligence; Representation (politics); Adaptation (eye); Key (lock); Sentiment analysis; Natural language; Translation (biology); Parallel corpora; Noise (video); Machine learning","score_opus":0.03508279994780869,"score_gpt":0.3290336885755691,"score_spread":0.2939508886277604,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2120158336","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9708086,0.00065400713,0.015990905,0.00038608967,0.00044940313,0.00052409933,0.0011032313,0.0026482174,0.007435437],"genre_scores_gemma":[0.94997513,0.0002993061,0.034721702,0.00073338783,0.00016363703,0.00055788044,0.005980571,0.0006424595,0.006926008],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.994584,0.0029752045,0.0004172167,0.0011147482,0.0007202533,0.00018854167],"domain_scores_gemma":[0.98784864,0.0062381006,0.00039860685,0.0031899866,0.001925685,0.00039897594],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00595533,0.00091273896,0.0008915361,0.00078980613,0.00079651695,0.0007827425,0.0010108338,0.0009300165,0.002240641],"category_scores_gemma":[0.018930033,0.00033934208,0.00063846534,0.0012447278,0.00058158557,0.0016813084,0.001779207,0.0014283995,0.0016706613],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0074957274,0.007850026,0.05790671,0.0012181761,0.0014148876,0.0019145674,0.003424327,0.021656772,0.13497415,0.000930304,0.032332093,0.72888225],"study_design_scores_gemma":[0.0018986722,0.011829093,0.24701603,0.00024571523,0.0019571441,0.0040140417,0.0056116106,0.4795981,0.17899941,0.006074841,0.06212859,0.0006268296],"about_ca_topic_score_codex":0.0038076907,"about_ca_topic_score_gemma":0.003452775,"teacher_disagreement_score":0.00595533,"about_ca_system_score_codex":0.00036022603,"about_ca_system_score_gemma":0.00044254973,"threshold_uncertainty_score":0.031495154},"labels":[],"label_agreement":null},{"id":"W2124618303","doi":"","title":"Substring-Based Transliteration","year":2007,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Transliteration; Substring; Computer science; Margin (machine learning); Artificial intelligence; Word (group theory); Natural language processing; Machine translation; Speech recognition; Programming language; Machine learning; Data structure; Mathematics","score_opus":0.011120757289096535,"score_gpt":0.27550062212974663,"score_spread":0.26437986484065007,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2124618303","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00989526,0.00021458624,0.97913456,0.00019275234,0.00011342203,0.000068893736,0.0002115368,0.0040569734,0.0061119725],"genre_scores_gemma":[0.29259974,0.00048006402,0.6900888,0.00031042102,0.000096452626,0.00016968451,0.001151221,0.001358785,0.013744816],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99867404,0.0003185599,0.00011551832,0.00042234,0.00037779397,0.00009185897],"domain_scores_gemma":[0.9973017,0.0011431545,0.00015328554,0.00077299005,0.0005809411,0.000048051068],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008406619,0.0005936558,0.00077990995,0.00059186574,0.00056406256,0.0012589344,0.0015133035,0.00094375457,0.009057875],"category_scores_gemma":[0.004397772,0.0003364872,0.0006686663,0.0009635891,0.0008478089,0.0023829797,0.0012297847,0.0014868461,0.0061921887],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029293157,0.00018093486,0.0011563381,0.00082545285,0.00010602876,0.0004762609,0.00080566603,0.054993704,0.19509092,0.13800567,0.014618146,0.593448],"study_design_scores_gemma":[0.00005306216,0.0002872222,0.00055128866,0.000068978385,0.00009866714,0.0008506349,0.00017494347,0.50192326,0.30414367,0.100856446,0.09090339,0.000088565765],"about_ca_topic_score_codex":0.0007096136,"about_ca_topic_score_gemma":0.0008742723,"teacher_disagreement_score":0.009057875,"about_ca_system_score_codex":0.00053664966,"about_ca_system_score_gemma":0.0010239854,"threshold_uncertainty_score":0.03030163},"labels":[],"label_agreement":null},{"id":"W2146546639","doi":"","title":"Learning Bigrams from Unigrams","year":2008,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Bigram; Perplexity; Computer science; Language model; Natural language processing; Artificial intelligence; Word (group theory); Trigram; Oracle; Set (abstract data type); Speech recognition; Mathematics; Programming language","score_opus":0.022393896773708673,"score_gpt":0.24455504663165106,"score_spread":0.22216114985794239,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2146546639","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1088765,0.0011678617,0.881627,0.0004976408,0.00015788536,0.000094216804,0.0007399554,0.004354657,0.0024841614],"genre_scores_gemma":[0.64789975,0.0009393764,0.33947104,0.00031786438,0.00023041801,0.0002246472,0.0034314722,0.0006826097,0.0068028737],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990152,0.00038465482,0.000076205724,0.00027000738,0.00014191394,0.00011190582],"domain_scores_gemma":[0.99555343,0.0029371232,0.00027390893,0.0006914452,0.00041863663,0.00012553591],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017826995,0.0012680431,0.0010788671,0.0033352089,0.000702576,0.0013675707,0.0015675573,0.0013942198,0.0030891895],"category_scores_gemma":[0.010073748,0.0007294053,0.0010378715,0.0019878782,0.0006651078,0.004932363,0.001828956,0.0023234587,0.0029764713],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064482796,0.00027192305,0.0058782287,0.00050950045,0.00023047328,0.0006778223,0.0009754828,0.06455686,0.014482225,0.029298857,0.010209955,0.8722639],"study_design_scores_gemma":[0.000050266975,0.00016424165,0.0012470359,0.000082423016,0.00006265422,0.00030871233,0.00034008056,0.8785776,0.0089916615,0.10546125,0.004651956,0.000062133615],"about_ca_topic_score_codex":0.0016208283,"about_ca_topic_score_gemma":0.002692252,"teacher_disagreement_score":0.0033352089,"about_ca_system_score_codex":0.00054708414,"about_ca_system_score_gemma":0.0009781776,"threshold_uncertainty_score":0.0103343725},"labels":[],"label_agreement":null},{"id":"W2161427841","doi":"","title":"Summarizing Emails with Conversational Cohesion and Subjectivity","year":2008,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Automatic summarization; Cohesion (chemistry); Computer science; PageRank; Cosine similarity; Natural language processing; Artificial intelligence; Graph; Sentence; Empirical research; Information retrieval; Subjectivity; Theoretical computer science; Pattern recognition (psychology); Mathematics","score_opus":0.013367564952236293,"score_gpt":0.24368251737275914,"score_spread":0.23031495242052286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2161427841","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11608325,0.002279369,0.8766132,0.00068054063,0.000121475954,0.00021388222,0.00048423195,0.0013502054,0.0021737854],"genre_scores_gemma":[0.56726617,0.0012689265,0.42654225,0.00013882475,0.0004897352,0.00021023533,0.001941613,0.00025047362,0.0018918167],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968136,0.0015834243,0.00032057246,0.00059232773,0.0005747551,0.0001153062],"domain_scores_gemma":[0.9846705,0.010240631,0.0018839203,0.0010429437,0.0019325969,0.00022947125],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033245555,0.0012263142,0.0009969994,0.004025226,0.00084124925,0.0024901538,0.00075536635,0.0011198987,0.00093759777],"category_scores_gemma":[0.02397137,0.0005320183,0.00077306037,0.0024095858,0.0006215124,0.005029156,0.0013531175,0.0006785095,0.00053316937],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009966069,0.0002481605,0.01733918,0.0027270676,0.00060651044,0.00088239013,0.006635359,0.065304674,0.04592604,0.024228247,0.0073124743,0.82779336],"study_design_scores_gemma":[0.000092751754,0.0007253079,0.020581605,0.00031306554,0.0010919403,0.00094156736,0.0042529064,0.7659903,0.04293973,0.13874528,0.024121353,0.00020413278],"about_ca_topic_score_codex":0.0010799741,"about_ca_topic_score_gemma":0.0015033331,"teacher_disagreement_score":0.004025226,"about_ca_system_score_codex":0.0005924078,"about_ca_system_score_gemma":0.0006183968,"threshold_uncertainty_score":0.017582119},"labels":[],"label_agreement":null},{"id":"W2166202273","doi":"","title":"Bootstrapping a Stochastic Transducer for Arabic-English Transliteration Extraction","year":2007,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Bootstrapping (finance); Computer science; Transducer; Scripting language; Artificial intelligence; Natural language processing; Metric (unit); Transliteration; Task (project management); Speech recognition; Similarity (geometry); Arabic; Pattern recognition (psychology); Programming language; Linguistics; Acoustics; Engineering; Mathematics","score_opus":0.014098183086053236,"score_gpt":0.2948093699806726,"score_spread":0.2807111868946193,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166202273","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047072686,0.000107674845,0.94682884,0.00030352827,0.00006859448,0.0000640031,0.00021193104,0.0043966672,0.0009460948],"genre_scores_gemma":[0.63805217,0.00011862488,0.3563839,0.00031698906,0.000093244984,0.000271262,0.0015461756,0.0003394151,0.002878192],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989686,0.00037671614,0.00006543864,0.00028745402,0.00021425955,0.00008745825],"domain_scores_gemma":[0.996691,0.0022547306,0.00013747632,0.0002654483,0.00055128516,0.00009999697],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001306289,0.00069780764,0.00083025,0.00066302216,0.0004962951,0.000598632,0.0011350202,0.0010299319,0.0026476888],"category_scores_gemma":[0.006989904,0.00042091834,0.00068795244,0.0005384317,0.0005704576,0.0012944487,0.0012572567,0.001568389,0.002431586],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058879476,0.00028494856,0.0045080297,0.00020808469,0.00011220447,0.00049985806,0.00051595946,0.12273517,0.101019904,0.008974833,0.0064389626,0.75411326],"study_design_scores_gemma":[0.000011386517,0.00009159778,0.0005117472,0.00000664707,0.0000145657295,0.00010961638,0.00003794138,0.9742848,0.019387374,0.0043167663,0.0012104389,0.000016993632],"about_ca_topic_score_codex":0.0022320377,"about_ca_topic_score_gemma":0.0037447782,"teacher_disagreement_score":0.0026476888,"about_ca_system_score_codex":0.00039810245,"about_ca_system_score_gemma":0.0010772066,"threshold_uncertainty_score":0.008857429},"labels":[],"label_agreement":null},{"id":"W2250341893","doi":"","title":"Mapping Source to Target Strings without Alignment by Analogical Learning: A Case Study with Transliteration","year":2013,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Transliteration; Computer science; Natural language processing; Phrase; Analogy; Artificial intelligence; Machine translation; Task (project management); Translation (biology); Rule-based machine translation; Machine learning; Linguistics; Engineering","score_opus":0.010253602251457633,"score_gpt":0.2590147617676204,"score_spread":0.24876115951616276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250341893","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9000833,0.0009240702,0.081973985,0.0016989292,0.000062237625,0.00037674367,0.00063251227,0.0011285197,0.013119727],"genre_scores_gemma":[0.9013769,0.00035129217,0.093037456,0.00028616586,0.000048210146,0.00016554572,0.0005800484,0.0003292841,0.0038250575],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99547195,0.0030847236,0.0003358724,0.0004429277,0.0004989469,0.00016557188],"domain_scores_gemma":[0.97325486,0.021031415,0.00072091556,0.0035806003,0.0011937316,0.00021840126],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003982951,0.00083069113,0.0009803104,0.0010277838,0.0015777225,0.0018196797,0.0020147236,0.003441906,0.0040131756],"category_scores_gemma":[0.028795637,0.00042885123,0.00079471373,0.002504324,0.0016085582,0.004666582,0.0019127923,0.0020967417,0.0014836256],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026630391,0.0055127256,0.035652332,0.0036493833,0.00040933097,0.037164845,0.03544879,0.09575075,0.03850665,0.030721594,0.015833003,0.69868755],"study_design_scores_gemma":[0.0017565096,0.0043617054,0.025584053,0.00041311592,0.0005218175,0.029713076,0.023825599,0.5814965,0.15688834,0.08422687,0.090884194,0.000328216],"about_ca_topic_score_codex":0.0037373707,"about_ca_topic_score_gemma":0.0053921933,"teacher_disagreement_score":0.0040131756,"about_ca_system_score_codex":0.0007223265,"about_ca_system_score_gemma":0.00054283394,"threshold_uncertainty_score":0.021064103},"labels":[],"label_agreement":null},{"id":"W2250669704","doi":"","title":"Probabilistic Domain Modelling With Contextualized Distributional Semantic Vectors","year":2013,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Automatic summarization; Computer science; Distributional semantics; Artificial intelligence; Probabilistic logic; Natural language processing; Domain (mathematical analysis); Semantics (computer science); Generative grammar; Generative model; Hidden Markov model; Semantic space; Machine learning; Semantic similarity; Mathematics","score_opus":0.01520811403659151,"score_gpt":0.23146867420578987,"score_spread":0.21626056016919837,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250669704","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007571125,0.00012845221,0.9898398,0.00017063208,0.000029020108,0.000044824923,0.00031459288,0.0006781381,0.0012234147],"genre_scores_gemma":[0.444993,0.0005714953,0.5437826,0.00021749867,0.0001460398,0.00042374505,0.0037635474,0.0004956953,0.0056064287],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986442,0.000603499,0.000084420564,0.000385546,0.00019933764,0.00008312689],"domain_scores_gemma":[0.9969458,0.001975573,0.00020771474,0.00047115627,0.00032075722,0.00007901756],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017737547,0.00070363673,0.0006232508,0.0015632474,0.000508959,0.0016041809,0.0018844877,0.0010479225,0.0032828178],"category_scores_gemma":[0.006248565,0.00058506994,0.0014262212,0.0018680362,0.0007194322,0.0033144355,0.0016312297,0.0019098513,0.0015155532],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019273399,0.00015056673,0.0029732094,0.00026465362,0.00014487238,0.00029207394,0.00071454915,0.5781695,0.005690348,0.2160835,0.0074942303,0.18782976],"study_design_scores_gemma":[0.000011477206,0.000018068013,0.00023428767,0.00001482022,0.000015845433,0.000058856476,0.000046221656,0.91741914,0.0012502226,0.077647336,0.0032683017,0.000015398002],"about_ca_topic_score_codex":0.0026657472,"about_ca_topic_score_gemma":0.005944633,"teacher_disagreement_score":0.0032828178,"about_ca_system_score_codex":0.0008596629,"about_ca_system_score_gemma":0.0010589861,"threshold_uncertainty_score":0.010982096},"labels":[],"label_agreement":null},{"id":"W3036997889","doi":"","title":"Using Attention-based Bidirectional LSTM to Identify Different Categories of Offensive Language Directed Toward Female Celebrities","year":2019,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Offensive; Harassment; Perspective (graphical); Margin (machine learning); Computer science; Social media; Artificial intelligence; Natural language processing; Psychology; Machine learning; Social psychology; World Wide Web; Mathematics","score_opus":0.02115456455403126,"score_gpt":0.2868773790427908,"score_spread":0.26572281448875956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3036997889","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6844278,0.0035624763,0.27646178,0.0021447048,0.001198644,0.0002592402,0.0053731077,0.010034158,0.01653807],"genre_scores_gemma":[0.94774604,0.00048988,0.037874665,0.00040576857,0.00011961005,0.0001563551,0.0042848284,0.00011252656,0.008810393],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997776,0.000040380215,0.000012824321,0.00008381894,0.00002889702,0.000056470188],"domain_scores_gemma":[0.9995844,0.00019028402,0.000039180883,0.000032071344,0.00012787145,0.000026219619],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047856953,0.0012188845,0.0004863312,0.0007437851,0.00031058476,0.000676792,0.0008773964,0.0008600107,0.0018865804],"category_scores_gemma":[0.001452094,0.00028455837,0.00053511193,0.00071331224,0.00029794427,0.0010282951,0.00084693485,0.0012762703,0.0011440227],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010294454,0.00080217764,0.017726075,0.00050695054,0.00034022974,0.00063400157,0.00073349976,0.15324014,0.06715653,0.0022546912,0.019169718,0.73640656],"study_design_scores_gemma":[0.00001762234,0.00011394839,0.0034336578,0.000043127722,0.00006967279,0.000078102996,0.00013374937,0.9839684,0.008584346,0.0018893621,0.0016438153,0.000024245166],"about_ca_topic_score_codex":0.013070194,"about_ca_topic_score_gemma":0.017734371,"teacher_disagreement_score":0.013070194,"about_ca_system_score_codex":0.00069249154,"about_ca_system_score_gemma":0.00069972384,"threshold_uncertainty_score":0.025988221},"labels":[],"label_agreement":null},{"id":"W3037566932","doi":"","title":"Augmenting Named Entity Recognition with Commonsense Knowledge","year":2019,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Université du Québec","funders":"","keywords":"Commonsense knowledge; Computer science; Natural language processing; Artificial intelligence; Named-entity recognition; Ambiguity; Knowledge base; Natural language understanding; Natural language; Word embedding; Question answering; Embedding; Task (project management)","score_opus":0.019435820131858655,"score_gpt":0.25008630713282126,"score_spread":0.2306504870009626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037566932","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06969673,0.0011588283,0.90508664,0.00068947766,0.00023229928,0.0001312904,0.0013607218,0.014750937,0.006893104],"genre_scores_gemma":[0.6793693,0.0006678288,0.30787987,0.00033030694,0.00012270981,0.00008289336,0.0063067726,0.0003832106,0.0048571546],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987594,0.0003097872,0.000090046924,0.00048682065,0.0002489571,0.000104988976],"domain_scores_gemma":[0.9962058,0.0016359396,0.00023218161,0.0013554269,0.00048640676,0.00008420384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021089963,0.0008928347,0.0009279219,0.0030153322,0.00067170564,0.0015091124,0.0016784399,0.0013486659,0.0036639345],"category_scores_gemma":[0.0057560806,0.0004773476,0.0010604181,0.0020458146,0.0008017191,0.010584362,0.0032998163,0.0018702648,0.0028922546],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023534073,0.00031393295,0.0045415945,0.0004171685,0.00022369865,0.00064114766,0.0005600667,0.046921395,0.024956662,0.011744256,0.011484149,0.8979607],"study_design_scores_gemma":[0.000019832376,0.0001264048,0.0046038474,0.00010188364,0.00016198984,0.0007415318,0.0003309118,0.8833186,0.039865803,0.045203224,0.025430754,0.00009519204],"about_ca_topic_score_codex":0.0027464477,"about_ca_topic_score_gemma":0.0064456863,"teacher_disagreement_score":0.0036639345,"about_ca_system_score_codex":0.00043626598,"about_ca_system_score_gemma":0.00071599387,"threshold_uncertainty_score":0.01225704},"labels":[],"label_agreement":null},{"id":"W3088355435","doi":"","title":"PACTE: A colloaborative platform for textual annotation","year":2017,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; École de Technologie Supérieure","funders":"","keywords":"Annotation; Computer science; Natural language processing; Artificial intelligence; Information retrieval; World Wide Web","score_opus":0.021316635145724953,"score_gpt":0.31386733643635356,"score_spread":0.2925507012906286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3088355435","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004116265,0.0003063716,0.7559756,0.0010190788,0.0009929384,0.00088864705,0.037135758,0.17682534,0.022740033],"genre_scores_gemma":[0.061878968,0.00053053506,0.6996862,0.0011725603,0.00064587785,0.0030739687,0.14652331,0.03994776,0.0465408],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9959992,0.0012128222,0.00048164767,0.00091347424,0.0011663664,0.00022650884],"domain_scores_gemma":[0.98639363,0.005316324,0.000699389,0.0041597863,0.0025709353,0.000859965],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004145683,0.002620617,0.0013358493,0.0056023807,0.0030371188,0.004033516,0.0035881395,0.0025858618,0.07197965],"category_scores_gemma":[0.017119762,0.0016698646,0.0014128542,0.004223833,0.001462979,0.010915282,0.011706512,0.0034491317,0.05544557],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017932795,0.0003818393,0.0017097377,0.0025522243,0.00019177621,0.0016269467,0.0043280134,0.00276745,0.049995255,0.082830794,0.6072373,0.24458542],"study_design_scores_gemma":[0.0001719643,0.00012944128,0.001226023,0.00026888345,0.000097540025,0.00066037383,0.0010595107,0.050408784,0.032321356,0.054008417,0.8594098,0.00023784433],"about_ca_topic_score_codex":0.005808033,"about_ca_topic_score_gemma":0.008825713,"teacher_disagreement_score":0.07197965,"about_ca_system_score_codex":0.00097249245,"about_ca_system_score_gemma":0.0035592725,"threshold_uncertainty_score":0.24079591},"labels":[],"label_agreement":null},{"id":"W3177369674","doi":"","title":"AdElectra - Effective Adversarial Electra","year":2021,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Classical Philosophy and Thought","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Adversarial system; Computer science; Artificial intelligence","score_opus":0.014771906817406777,"score_gpt":0.23645921032382644,"score_spread":0.22168730350641966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3177369674","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010989602,0.00046584167,0.95199835,0.0013856784,0.000309904,0.000059887767,0.00023427115,0.00046788348,0.03408861],"genre_scores_gemma":[0.6876865,0.00096430036,0.19569202,0.0011942519,0.00048541525,0.00027142293,0.00077701523,0.0005269053,0.11240221],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99921954,0.0003008534,0.000031002557,0.00016484431,0.00018766621,0.00009612643],"domain_scores_gemma":[0.99798536,0.0013117611,0.000060042395,0.0002962702,0.0002435895,0.00010283751],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016107477,0.0011061887,0.00093571516,0.00074353564,0.00084116706,0.0015321121,0.0015896177,0.0017482393,0.019389879],"category_scores_gemma":[0.005482165,0.0004891922,0.0006663745,0.0005785575,0.0017588448,0.0026230973,0.004255781,0.002791852,0.0027185564],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013553626,0.000082251994,0.00028696161,0.00012232484,0.00004923632,0.00014892392,0.00007825236,0.31400824,0.0019304462,0.5868081,0.014080666,0.082269154],"study_design_scores_gemma":[0.000012742127,0.000034467026,0.000057321336,0.000018968485,0.000008056448,0.000067777335,0.000015708938,0.69878167,0.00081794726,0.29568487,0.004491941,0.000008458431],"about_ca_topic_score_codex":0.0010588049,"about_ca_topic_score_gemma":0.0017841078,"teacher_disagreement_score":0.019389879,"about_ca_system_score_codex":0.0010613317,"about_ca_system_score_gemma":0.0010085895,"threshold_uncertainty_score":0.06486559},"labels":[],"label_agreement":null},{"id":"W3197573382","doi":"","title":"Proceedings of the ACL-IJCNLP 2021 Student Research Workshop.","year":2021,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Dental Research and COVID-19","field":"Dentistry","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Mathematics education; Data science; Psychology","score_opus":0.07961984896349812,"score_gpt":0.418582325272663,"score_spread":0.3389624763091649,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3197573382","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014027575,0.045514494,0.1785391,0.17980596,0.17486298,0.0015356981,0.099001445,0.033893254,0.27281952],"genre_scores_gemma":[0.034732185,0.016646251,0.084691904,0.017035708,0.020700544,0.0022290796,0.2930169,0.01117715,0.5197702],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99218494,0.004252728,0.0005716434,0.0010641422,0.001432854,0.0004937687],"domain_scores_gemma":[0.97860044,0.00781996,0.00037382907,0.0027968574,0.0074989824,0.0029098848],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012352221,0.0024114018,0.0030387538,0.0029718378,0.0036619161,0.012180718,0.0045533525,0.0048669158,0.16829245],"category_scores_gemma":[0.02253159,0.0017039307,0.0018951521,0.002576815,0.0017899633,0.013485069,0.0072217775,0.006280842,0.14050281],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000106622734,0.00008959221,0.00013848206,0.00012998632,0.000019125922,0.00005403465,0.00011317916,0.00008587656,0.00023435369,0.0014604795,0.97814184,0.019426359],"study_design_scores_gemma":[0.00010196322,0.000041436866,0.0008452629,0.00039512402,0.000057791578,0.000247732,0.0007672828,0.0020304695,0.0010247291,0.0083793625,0.9860596,0.0000492135],"about_ca_topic_score_codex":0.018765898,"about_ca_topic_score_gemma":0.03657106,"teacher_disagreement_score":0.16829245,"about_ca_system_score_codex":0.003054442,"about_ca_system_score_gemma":0.004882525,"threshold_uncertainty_score":0.56299436},"labels":[],"label_agreement":null},{"id":"W4531093","doi":"","title":"Extraction of Disease-Treatment Semantic Relations from Biomedical Sentences","year":2010,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Relationship extraction; Relation (database); Semantic relation; Measure (data warehouse); Computer science; Natural language processing; Focus (optics); Artificial intelligence; Semantics (computer science); Information retrieval; Information extraction; Data mining; Medicine; Programming language","score_opus":0.012831567719147929,"score_gpt":0.2901257257106199,"score_spread":0.27729415799147195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4531093","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23160559,0.012894242,0.60674894,0.007939966,0.0015537458,0.0030587341,0.10171468,0.008703025,0.025781082],"genre_scores_gemma":[0.3063249,0.0025894954,0.5997563,0.00081711495,0.0006653175,0.0008578229,0.086673215,0.00029774423,0.0020179912],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970175,0.0010481251,0.0005859991,0.0006483123,0.00059765665,0.000102406164],"domain_scores_gemma":[0.9887217,0.007994466,0.0012139634,0.0006097175,0.001276883,0.00018317725],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002777044,0.0018594894,0.0009872274,0.0076245246,0.0013546454,0.0015557129,0.0007805753,0.0014487085,0.0045221425],"category_scores_gemma":[0.012331245,0.0005195304,0.0017803541,0.0037628154,0.00055676274,0.0028066065,0.0013160845,0.0015192045,0.002280975],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014800844,0.0010691182,0.033326,0.0104933735,0.0007212496,0.005336471,0.0041742725,0.0097421175,0.1425313,0.033111703,0.045336932,0.7126773],"study_design_scores_gemma":[0.000577191,0.0012612836,0.11265064,0.0025095239,0.0031379326,0.013308503,0.0066016647,0.21402572,0.17376798,0.1312646,0.3404145,0.0004804946],"about_ca_topic_score_codex":0.0019997214,"about_ca_topic_score_gemma":0.0022639614,"teacher_disagreement_score":0.0076245246,"about_ca_system_score_codex":0.0009379207,"about_ca_system_score_gemma":0.002812756,"threshold_uncertainty_score":0.015128076},"labels":[],"label_agreement":null}]}