{"meta":{"query_hash":"a6c2b8c45c44","filters":{"venue":"Information Retrieval"},"cohort_total":26,"direct_labels_cover":0,"predictions_cover":26,"exported":26,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/a6c2b8c45c44","api":"https://metacan.xera.ac/api/v1/cohort?venue=Information+Retrieval"},"results":[{"id":"W1503824486","doi":"10.1023/a:1023936321956","title":"Query Expansion with Long-Span Collocates","year":2003,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":53,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Microsoft Research","keywords":"Collocation (remote sensing); Computer science; Window (computing); Information retrieval; Mathematics; Machine learning","score_opus":0.009880289279499396,"score_gpt":0.22393088713362766,"score_spread":0.21405059785412828,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1503824486","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39722595,0.0044198986,0.57217276,0.0012876341,0.00035455657,0.0005837611,0.0021233212,0.0065776645,0.0152545115],"genre_scores_gemma":[0.8546096,0.000704416,0.13406035,0.00024697458,0.00017580825,0.00022394837,0.0026638363,0.00038160986,0.0069333836],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99824524,0.00067711266,0.00012702239,0.00034841226,0.00045038626,0.00015191906],"domain_scores_gemma":[0.99389595,0.003472231,0.00023980944,0.0012611866,0.00096952135,0.0001612446],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015612198,0.00072022516,0.0010145797,0.0020133608,0.00082036894,0.00086285814,0.0010075513,0.0009309647,0.008164558],"category_scores_gemma":[0.012400556,0.0004386032,0.00053249113,0.0032243272,0.0005407943,0.004916726,0.0016883478,0.00085520605,0.0023187601],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003545396,0.0015805144,0.009916513,0.0010097865,0.00028846288,0.0011712031,0.0015340684,0.08860492,0.09957104,0.026132874,0.04911566,0.7175297],"study_design_scores_gemma":[0.00021587666,0.0005539703,0.004741539,0.00004920645,0.00019976773,0.0009767666,0.00053982524,0.92178565,0.028564198,0.024731494,0.017548436,0.00009317265],"about_ca_topic_score_codex":0.00505371,"about_ca_topic_score_gemma":0.007730771,"teacher_disagreement_score":0.008164558,"about_ca_system_score_codex":0.00058076496,"about_ca_system_score_gemma":0.0009312957,"threshold_uncertainty_score":0.027313173},"labels":[],"label_agreement":null},{"id":"W1685426458","doi":"10.1007/s10791-011-9162-z","title":"Efficient and effective spam filtering and re-ranking for large web datasets","year":2011,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":277,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Institute of Standards and Technology","keywords":"Computer science; Information retrieval; Ranking (information retrieval); Relevance feedback; Relevance (law); Search engine; Learning to rank; Set (abstract data type); Honeypot; Web page; Rank (graph theory); Data mining; World Wide Web; Artificial intelligence; Image retrieval; Mathematics","score_opus":0.015493414875478324,"score_gpt":0.23253805024293472,"score_spread":0.2170446353674564,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1685426458","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1723719,0.007815761,0.7662311,0.0030536314,0.00076596363,0.0009342374,0.007597427,0.037602674,0.0036273114],"genre_scores_gemma":[0.28541893,0.0011334067,0.69163984,0.0004526824,0.000863334,0.00038197832,0.013402135,0.00062267616,0.0060849553],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9945162,0.0016903838,0.00053961325,0.0008355431,0.0018798205,0.0005383738],"domain_scores_gemma":[0.9875721,0.004589205,0.0008445789,0.0038900143,0.0026855064,0.0004186419],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004885714,0.0019512242,0.0035049904,0.008493712,0.0019826046,0.0034920704,0.002250577,0.0025853454,0.0021984545],"category_scores_gemma":[0.016443964,0.0010469217,0.0019011988,0.0060124374,0.0007814644,0.005201903,0.001972395,0.0016633294,0.003654731],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005535228,0.0008198719,0.009108506,0.0005532436,0.00041305364,0.0002274252,0.00019150136,0.031477183,0.028037775,0.0025562095,0.071518615,0.85454303],"study_design_scores_gemma":[0.00014679482,0.00022212949,0.0075618364,0.00003258008,0.00020402396,0.0005167365,0.00023990544,0.9496571,0.018263439,0.013872406,0.009207196,0.00007591258],"about_ca_topic_score_codex":0.007594257,"about_ca_topic_score_gemma":0.019639008,"teacher_disagreement_score":0.008493712,"about_ca_system_score_codex":0.0012694346,"about_ca_system_score_gemma":0.003098424,"threshold_uncertainty_score":0.025838435},"labels":[],"label_agreement":null},{"id":"W2000952055","doi":"10.1007/s10791-006-9397-2","title":"Introduction to the special issue on the 27th European Conference on Information Retrieval Research","year":2006,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Waterloo","keywords":"Information retrieval; Political science; Library science; Data science; Computer science","score_opus":0.03850458048473922,"score_gpt":0.29410519826420334,"score_spread":0.25560061777946413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2000952055","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006136336,0.08643523,0.0087329475,0.038013525,0.81171125,0.0002604088,0.0012134879,0.0009673828,0.052052103],"genre_scores_gemma":[0.003374342,0.061662666,0.004218692,0.019362845,0.5341825,0.00040972908,0.0033835259,0.0019115354,0.3714942],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9956245,0.0006161695,0.00044633937,0.00060364313,0.0022052173,0.0005041853],"domain_scores_gemma":[0.9838923,0.0036245077,0.0009606994,0.0013607701,0.0068959487,0.0032658125],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059752157,0.0028586523,0.008110515,0.0099935,0.0027438304,0.0126855485,0.0031166242,0.0054334975,0.19469276],"category_scores_gemma":[0.010976057,0.0010022388,0.0024792252,0.005816402,0.0013458444,0.009754769,0.0042877253,0.0066762543,0.15660852],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000045725734,0.00003445265,0.000082781,0.00019703123,0.000013918297,0.000028466475,0.0000124435555,0.00003241692,0.00021837841,0.0005784182,0.9652282,0.03352786],"study_design_scores_gemma":[0.000018876013,0.000059373502,0.00064290117,0.00022675867,0.000035546775,0.00013764657,0.000033897853,0.00020413526,0.00016064192,0.0013234288,0.9971329,0.00002393123],"about_ca_topic_score_codex":0.0015039924,"about_ca_topic_score_gemma":0.003562294,"teacher_disagreement_score":0.19469276,"about_ca_system_score_codex":0.002583321,"about_ca_system_score_gemma":0.0032498213,"threshold_uncertainty_score":0.6513122},"labels":[],"label_agreement":null},{"id":"W2008283248","doi":"10.1007/s10791-013-9220-9","title":"Latent word context model for information retrieval","year":2013,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Topic Modeling","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Ministry of Education, Culture, Sports, Science and Technology","keywords":"Computer science; Latent Dirichlet allocation; Word (group theory); Context (archaeology); Information retrieval; Natural language processing; Latent semantic analysis; Search engine indexing; Topic model; Relevance (law); Artificial intelligence; Probabilistic latent semantic analysis; Linguistics","score_opus":0.02483661079381079,"score_gpt":0.23769802741019738,"score_spread":0.21286141661638658,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2008283248","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010183527,0.0063549,0.97846425,0.00095682626,0.00020232525,0.00008465172,0.0009955954,0.0011051462,0.0016527383],"genre_scores_gemma":[0.5755696,0.009994214,0.38523626,0.0007698655,0.0012872367,0.0010599986,0.006459353,0.0006990386,0.01892441],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99740404,0.0014005876,0.00017201081,0.0004671043,0.00037473213,0.00018140729],"domain_scores_gemma":[0.99616915,0.0026517718,0.00023809951,0.00047034718,0.0003890096,0.00008161241],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028986835,0.0010552829,0.0022142555,0.0025204218,0.0007558711,0.002495525,0.0024519907,0.0022190518,0.004332417],"category_scores_gemma":[0.011401789,0.00077275647,0.0016735547,0.0043547694,0.0009203355,0.0058719846,0.0012929911,0.0029379912,0.0031407601],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008004137,0.00050865,0.0032404286,0.0012709951,0.0006560082,0.00037715433,0.00078132545,0.2073346,0.0076058484,0.39307812,0.02753184,0.3568146],"study_design_scores_gemma":[0.00004969746,0.00005722413,0.0006815299,0.00005740674,0.00013578428,0.00010273981,0.00005152856,0.80075395,0.0008455367,0.19104514,0.006172834,0.00004664924],"about_ca_topic_score_codex":0.0075179352,"about_ca_topic_score_gemma":0.0075265607,"teacher_disagreement_score":0.0075179352,"about_ca_system_score_codex":0.0014092543,"about_ca_system_score_gemma":0.0016894806,"threshold_uncertainty_score":0.015329838},"labels":[],"label_agreement":null},{"id":"W2032328503","doi":"10.1007/s10791-009-9108-x","title":"Document clustering of scientific texts using citation contexts","year":2009,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Cluster analysis; Information retrieval; Computer science; Document clustering; Citation; Vocabulary; Context (archaeology); Similarity (geometry); Representation (politics); Document retrieval; Natural language processing; Artificial intelligence; World Wide Web; Linguistics","score_opus":0.016025685924846902,"score_gpt":0.2950436954188471,"score_spread":0.27901800949400024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2032328503","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43528467,0.02988485,0.488127,0.002138877,0.0019418302,0.0011894114,0.009118278,0.005384562,0.026930515],"genre_scores_gemma":[0.6016658,0.006460908,0.36630785,0.00015246746,0.0016504092,0.0005931752,0.01205742,0.0006688847,0.010443041],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978617,0.00055193325,0.00030844708,0.00042824374,0.00069155026,0.00015813031],"domain_scores_gemma":[0.99230534,0.0035796973,0.00060034555,0.0005478903,0.0026807154,0.0002859683],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0016699878,0.0008086043,0.0011066889,0.02648126,0.0024123208,0.0040888195,0.0010495521,0.0011271278,0.0031576708],"category_scores_gemma":[0.011847319,0.00038917956,0.0012473554,0.022339862,0.00064202555,0.0024870087,0.0012086096,0.0009756474,0.0020975948],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008256128,0.00041090476,0.015022291,0.0016576733,0.00042566075,0.00037108985,0.0013423082,0.010132029,0.037552208,0.013687033,0.017721085,0.900852],"study_design_scores_gemma":[0.00044168904,0.0011516517,0.10011501,0.0012248359,0.003434505,0.0022770888,0.0038670625,0.50720596,0.103658564,0.12631258,0.14975165,0.00055942923],"about_ca_topic_score_codex":0.00395782,"about_ca_topic_score_gemma":0.008434736,"teacher_disagreement_score":0.9735187,"about_ca_system_score_codex":0.0010436877,"about_ca_system_score_gemma":0.002374801,"threshold_uncertainty_score":0.010563493},"labels":[],"label_agreement":null},{"id":"W2033681837","doi":"10.1007/s10791-011-9163-y","title":"Improving document clustering using Okapi BM25 feature weighting","year":2011,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Image Retrieval and Classification Techniques","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Weighting; Cluster analysis; Computer science; Artificial intelligence; Data mining; Feature (linguistics); Document clustering; Pattern recognition (psychology); Term (time)","score_opus":0.024820470294881734,"score_gpt":0.24203731868756767,"score_spread":0.21721684839268593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2033681837","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04692125,0.004228389,0.93250895,0.00027109732,0.0007061816,0.000263758,0.0009773949,0.01039429,0.0037287741],"genre_scores_gemma":[0.22937305,0.0013673037,0.74976254,0.0003207954,0.0003474439,0.00028541524,0.005917789,0.0009853313,0.011640251],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99800736,0.0003271185,0.00017476527,0.00041822955,0.0008859279,0.00018659548],"domain_scores_gemma":[0.9984641,0.00021586471,0.0000707106,0.000296273,0.0008949515,0.000058130678],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012862147,0.0015230255,0.0021479751,0.0038227092,0.001504776,0.0014973013,0.0017975757,0.0013011359,0.0038305018],"category_scores_gemma":[0.0039123013,0.00039313815,0.0015796016,0.0052015996,0.0003445635,0.00181931,0.0010477344,0.0012716395,0.004966393],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033510363,0.0003208881,0.0009958755,0.00016762869,0.00016651195,0.000049950297,0.000058758527,0.011027958,0.03079706,0.0014111182,0.02052903,0.9341402],"study_design_scores_gemma":[0.00010690637,0.00032649317,0.0068206326,0.00005601645,0.0005100734,0.00044362922,0.00019249535,0.8821591,0.08027036,0.007953618,0.02102135,0.00013925348],"about_ca_topic_score_codex":0.016652355,"about_ca_topic_score_gemma":0.022151096,"teacher_disagreement_score":0.016652355,"about_ca_system_score_codex":0.0009573669,"about_ca_system_score_gemma":0.0015021574,"threshold_uncertainty_score":0.033110857},"labels":[],"label_agreement":null},{"id":"W2056081579","doi":"10.1007/s10791-007-9042-8","title":"Hybrid index maintenance for contiguous inverted lists","year":2008,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"University of Waterloo","keywords":"Merge (version control); Computer science; Inverted index; Information retrieval; Data mining; Search engine indexing","score_opus":0.01558255272543664,"score_gpt":0.22993730688226618,"score_spread":0.21435475415682953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056081579","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.115449294,0.0033665786,0.85963035,0.00047313765,0.0003775763,0.00035727723,0.0019125816,0.011709381,0.0067238086],"genre_scores_gemma":[0.33985665,0.00068728405,0.6459939,0.00026035908,0.00030997265,0.00030146312,0.0040514586,0.0008863189,0.0076525654],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99846065,0.00021310018,0.00020454398,0.00024527375,0.0007285208,0.00014799286],"domain_scores_gemma":[0.9921969,0.0018394054,0.00047861433,0.0035978209,0.0017180301,0.0001693323],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014459636,0.0005462507,0.0012440853,0.0038798258,0.0013295492,0.0022977097,0.0030126139,0.0009342733,0.004715884],"category_scores_gemma":[0.007664511,0.00063316827,0.0005964817,0.0057818247,0.0006857652,0.0052392767,0.002228166,0.00084223616,0.0018475034],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009543591,0.00033338374,0.0029622002,0.00039493796,0.00014564436,0.00020975467,0.00027629087,0.018350365,0.033902608,0.014766849,0.021392992,0.9063106],"study_design_scores_gemma":[0.00046439175,0.0011043934,0.0049557826,0.00015706141,0.0004976747,0.0019179395,0.00044600043,0.7295048,0.13187353,0.089670986,0.039198767,0.00020864008],"about_ca_topic_score_codex":0.0034357177,"about_ca_topic_score_gemma":0.0070060315,"teacher_disagreement_score":0.004715884,"about_ca_system_score_codex":0.0007723093,"about_ca_system_score_gemma":0.0015012975,"threshold_uncertainty_score":0.015776217},"labels":[],"label_agreement":null},{"id":"W2062626555","doi":"10.1007/s10791-009-9105-0","title":"Swapping documents and terms","year":2009,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Agricultural Research Development Agency","keywords":"Relevance feedback; Computer science; Information retrieval; Relevance (law); Query expansion; Data mining; Artificial intelligence; Image retrieval","score_opus":0.009581800700385145,"score_gpt":0.25307819252788394,"score_spread":0.2434963918274988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2062626555","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6505876,0.00506926,0.26371795,0.00303477,0.0020998379,0.001109968,0.0043368344,0.009796022,0.060247745],"genre_scores_gemma":[0.84016544,0.0014321676,0.123660736,0.0006602156,0.00032008102,0.00028390266,0.0032217563,0.00109768,0.029157942],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963199,0.0012800882,0.00028483485,0.00065238436,0.001145825,0.00031700984],"domain_scores_gemma":[0.98865837,0.0046427357,0.0003853435,0.0042303363,0.0016600753,0.00042319746],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002680653,0.00066347,0.0009477152,0.0027356383,0.000890645,0.003296962,0.0014780373,0.0012677943,0.021370914],"category_scores_gemma":[0.025937608,0.000512286,0.00082297274,0.004271405,0.0010029041,0.006873904,0.002299457,0.0012672878,0.008488594],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027063792,0.0009855072,0.008662329,0.0006480611,0.00017945244,0.00033775708,0.0011243237,0.003946127,0.10509749,0.03756563,0.01612337,0.82262355],"study_design_scores_gemma":[0.00070148055,0.0024888767,0.027958991,0.0002971346,0.0011009694,0.0050708936,0.0032914984,0.13587902,0.39731768,0.20021032,0.22529,0.0003931457],"about_ca_topic_score_codex":0.0009419817,"about_ca_topic_score_gemma":0.00092911714,"teacher_disagreement_score":0.021370914,"about_ca_system_score_codex":0.000513286,"about_ca_system_score_gemma":0.0011606391,"threshold_uncertainty_score":0.07149285},"labels":[],"label_agreement":null},{"id":"W2069744915","doi":"10.1007/s10791-012-9192-1","title":"Extended structural relevance framework: a framework for evaluating structured document retrieval","year":2012,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Relevance (law); Information retrieval; Redundancy (engineering); Probabilistic logic; Task (project management); Tree (set theory); Data mining; Range (aeronautics); Artificial intelligence; Mathematics","score_opus":0.03216634289591799,"score_gpt":0.3490464434591703,"score_spread":0.31688010056325233,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2069744915","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025721101,0.0023492777,0.9655446,0.00033685868,0.00007214854,0.00059111125,0.0007030567,0.0016265393,0.0030552966],"genre_scores_gemma":[0.42726555,0.0009449161,0.5663996,0.000131677,0.00026311513,0.00071788137,0.0012330583,0.0002882862,0.0027558864],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9888593,0.0045725494,0.0006744446,0.0008858722,0.004588476,0.00041937985],"domain_scores_gemma":[0.98421025,0.007205093,0.0013216602,0.002326783,0.004501326,0.00043476312],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014277954,0.001337994,0.002301575,0.008555544,0.0012511958,0.0034842305,0.0029664964,0.002008161,0.0052500335],"category_scores_gemma":[0.043480877,0.0005958507,0.0016052817,0.0042394446,0.0015235359,0.006541034,0.0020805844,0.0014048758,0.0014591806],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012197054,0.0007704065,0.010141334,0.0014473404,0.0006222494,0.00026008472,0.0007208939,0.08017334,0.025801726,0.1463499,0.011688605,0.72080433],"study_design_scores_gemma":[0.00016321363,0.0011783185,0.005173422,0.00015628051,0.00038741855,0.0004457876,0.00018211044,0.8297148,0.010716447,0.14295292,0.008770803,0.00015850467],"about_ca_topic_score_codex":0.005485985,"about_ca_topic_score_gemma":0.0058847414,"teacher_disagreement_score":0.014277954,"about_ca_system_score_codex":0.0019063033,"about_ca_system_score_gemma":0.00327874,"threshold_uncertainty_score":0.075509906},"labels":[],"label_agreement":null},{"id":"W2074280490","doi":"10.1007/s10791-011-9168-6","title":"A study of the integration of passage-, document-, and cluster-based information for re-ranking search results","year":2011,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Israel Science Foundation; McGill University","keywords":"Ranking (information retrieval); Computer science; Information retrieval; Context (archaeology); Representation (politics); Query expansion; Task (project management); Cluster (spacecraft); Document retrieval; Data mining; Information integration; Geography","score_opus":0.05817615587411926,"score_gpt":0.29534894858965705,"score_spread":0.2371727927155378,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2074280490","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.765665,0.0048655695,0.21510085,0.0008940519,0.00010671686,0.00070481177,0.0005224843,0.0011886442,0.010951918],"genre_scores_gemma":[0.8924393,0.0005293049,0.104682386,0.000055436005,0.00007401819,0.00009423339,0.00035519197,0.00016138885,0.0016087522],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99508566,0.0026050066,0.0002217867,0.00043299343,0.0014283871,0.00022618026],"domain_scores_gemma":[0.92943984,0.05667843,0.0022832337,0.003627575,0.0069952216,0.0009757788],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006305924,0.0005698484,0.0012154881,0.0051960363,0.0008662579,0.0026474546,0.0013294189,0.00090633216,0.0017264566],"category_scores_gemma":[0.057311255,0.00043868367,0.00086186465,0.0077755568,0.0009791002,0.0056120288,0.00066335645,0.0011737602,0.0003250126],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0040228395,0.0025409698,0.062298194,0.0012704019,0.0013333329,0.00030799952,0.003954783,0.04607515,0.034996554,0.041674823,0.006326998,0.7951979],"study_design_scores_gemma":[0.00016123799,0.0017623592,0.058841992,0.0000960419,0.0010912077,0.0004044502,0.0009706925,0.906197,0.013700088,0.011723967,0.004868258,0.00018262469],"about_ca_topic_score_codex":0.01719508,"about_ca_topic_score_gemma":0.015817197,"teacher_disagreement_score":0.01719508,"about_ca_system_score_codex":0.0016788354,"about_ca_system_score_gemma":0.0015741993,"threshold_uncertainty_score":0.03419},"labels":[],"label_agreement":null},{"id":"W2082055394","doi":"10.1007/s10791-012-9218-8","title":"Increasing evaluation sensitivity to diversity","year":2013,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Science Foundation","keywords":"Computer science; Diversity (politics); Context (archaeology); Information retrieval; Discriminative model; Measure (data warehouse); Data science; Data mining; Artificial intelligence; Geography","score_opus":0.023805444270144614,"score_gpt":0.25938339518383075,"score_spread":0.23557795091368613,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2082055394","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8302637,0.0026838437,0.13402033,0.003195961,0.00020066633,0.00022450276,0.00023063975,0.001389529,0.027790971],"genre_scores_gemma":[0.98875123,0.00015270541,0.009589763,0.00032257798,0.0000856514,0.000028206825,0.00006204025,0.00010014761,0.0009077214],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9497899,0.031087933,0.0024218403,0.00580089,0.00940295,0.0014965807],"domain_scores_gemma":[0.5014905,0.42741203,0.016025653,0.035299342,0.017128665,0.002643789],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.031865887,0.0007833574,0.0013948572,0.0028335534,0.0009257513,0.0046638744,0.0010711745,0.0025708778,0.003536755],"category_scores_gemma":[0.27054426,0.0007639112,0.0007706834,0.0017662428,0.0017000028,0.007153212,0.004030129,0.0027740705,0.0006886284],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0060396907,0.0013631124,0.17309979,0.0013151738,0.0016679573,0.0007646561,0.004143725,0.10222827,0.106795825,0.04872802,0.006792291,0.54706156],"study_design_scores_gemma":[0.00042340878,0.0032056682,0.15069605,0.0004257455,0.0015172622,0.00289762,0.0019588512,0.5333737,0.098949134,0.19241054,0.013722734,0.00041924205],"about_ca_topic_score_codex":0.0013317303,"about_ca_topic_score_gemma":0.00087849976,"teacher_disagreement_score":0.031865887,"about_ca_system_score_codex":0.002269134,"about_ca_system_score_gemma":0.0011437273,"threshold_uncertainty_score":0.16852492},"labels":[],"label_agreement":null},{"id":"W2084946528","doi":"10.1007/s10791-006-9020-6","title":"Knowledge-based query expansion to support scenario-specific retrieval of medical free text","year":2007,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":79,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Institutes of Health; McMaster University","keywords":"Computer science; Query expansion; Unified Medical Language System; Information retrieval; Query language; Testbed; Query optimization; Precision and recall; Exploit; Recall; Data mining; World Wide Web","score_opus":0.020538311628058734,"score_gpt":0.3008048304354492,"score_spread":0.28026651880739045,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2084946528","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09532925,0.001011649,0.856819,0.0018722401,0.00019841186,0.0010241999,0.005377531,0.026430693,0.011937067],"genre_scores_gemma":[0.46831605,0.0006388643,0.5130407,0.0006193874,0.00014551415,0.0005836184,0.012729787,0.00063886243,0.00328711],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99880934,0.00043616,0.00015769107,0.00018535949,0.00031923538,0.00009224516],"domain_scores_gemma":[0.9966466,0.002170275,0.00014005954,0.00034771184,0.00058284035,0.00011243212],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015976052,0.00064914336,0.00076642825,0.0027620764,0.00048705377,0.0011158787,0.0013263721,0.0011469394,0.00836336],"category_scores_gemma":[0.008567872,0.00031569955,0.00073128066,0.0017691213,0.00032029004,0.0023898704,0.0015273795,0.0006612243,0.00248776],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018154676,0.0009832729,0.006045635,0.0014991778,0.00029644638,0.0028388766,0.001911335,0.06046873,0.08260421,0.02281441,0.09860475,0.72011757],"study_design_scores_gemma":[0.00030968245,0.00021220907,0.0033526153,0.0001591405,0.0002594514,0.0018072606,0.00076803105,0.8807873,0.042344667,0.023469152,0.046420544,0.00010992402],"about_ca_topic_score_codex":0.0031996563,"about_ca_topic_score_gemma":0.004793975,"teacher_disagreement_score":0.00836336,"about_ca_system_score_codex":0.00063796237,"about_ca_system_score_gemma":0.0008615773,"threshold_uncertainty_score":0.027978241},"labels":[],"label_agreement":null},{"id":"W2121315872","doi":"10.1007/s10791-013-9230-7","title":"Discover hidden web properties by random walk on bipartite graph","year":2013,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"Natural Sciences and Engineering Research Council of Canada; State Key Laboratory of Novel Software Technology; Nanjing University","keywords":"Zipf's law; Crawling; Simple random sample; Random walk; Sampling (signal processing); Bipartite graph; Statistics; Computer science; Sample size determination; Sample (material); Bernoulli's principle; Population; Mathematics; Graph; Theoretical computer science","score_opus":0.008604713258223986,"score_gpt":0.18922231192531797,"score_spread":0.18061759866709398,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121315872","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.57189447,0.0008809852,0.41944996,0.00085792417,0.00004232299,0.000118363394,0.0019832775,0.0020104619,0.002762224],"genre_scores_gemma":[0.9553607,0.00030242783,0.04074347,0.00010502373,0.00006371925,0.000048213315,0.0019194987,0.00011219708,0.0013447974],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993382,0.0002151849,0.000033141314,0.0001819096,0.00013610517,0.00009545631],"domain_scores_gemma":[0.993468,0.004428194,0.00078725885,0.000676261,0.00038155637,0.000258807],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006936954,0.0006015337,0.0011915268,0.0050093904,0.0006803951,0.0012621016,0.0009171172,0.0014764674,0.001332494],"category_scores_gemma":[0.006454105,0.00071584235,0.0010270454,0.0024161132,0.00081021397,0.0035262892,0.000926211,0.001109065,0.0005484079],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020328194,0.0015715566,0.11353395,0.0014322394,0.0011047178,0.0018321504,0.0009285028,0.41285318,0.052859187,0.13257164,0.019014739,0.26026517],"study_design_scores_gemma":[0.000026576246,0.000045028937,0.002637314,0.000014460408,0.00004700684,0.00011097181,0.000049974427,0.9417186,0.001270377,0.053624578,0.00044202385,0.00001308435],"about_ca_topic_score_codex":0.0032797207,"about_ca_topic_score_gemma":0.005123858,"teacher_disagreement_score":0.0050093904,"about_ca_system_score_codex":0.00065403746,"about_ca_system_score_gemma":0.00056458084,"threshold_uncertainty_score":0.006521225},"labels":[],"label_agreement":null},{"id":"W2141520705","doi":"10.1023/a:1026028229881","title":"Applying Machine Learning to Text Segmentation for Information Retrieval","year":2003,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Topic Modeling","field":"Computer Science","cited_by":76,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Segmentation; Computer science; Text segmentation; Artificial intelligence; Pattern recognition (psychology); Natural language processing; Word (group theory); Mathematics","score_opus":0.016645412208762866,"score_gpt":0.25568494074736936,"score_spread":0.2390395285386065,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141520705","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016460698,0.0018568216,0.9744391,0.00042954506,0.00016642034,0.00019143653,0.00023332499,0.004581825,0.0016408168],"genre_scores_gemma":[0.19853266,0.0016855876,0.7919687,0.0002758656,0.00051196327,0.00032763428,0.0017109981,0.0008884883,0.0040980643],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99805945,0.0009328028,0.0001661074,0.0003937302,0.00031570293,0.00013220888],"domain_scores_gemma":[0.994518,0.0039974777,0.0002636971,0.00048577698,0.0006177155,0.000117413045],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023408048,0.0011852646,0.0016790185,0.004115161,0.0011106795,0.0025003417,0.001268672,0.0015176508,0.0031278236],"category_scores_gemma":[0.00960903,0.00071805646,0.0015518145,0.003868007,0.00088751526,0.0032651532,0.0010387112,0.0017421303,0.0029929597],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033933713,0.00036011872,0.0018411519,0.000602051,0.0002307993,0.00013779118,0.00042240272,0.05574257,0.038491827,0.0083670085,0.011189953,0.8822751],"study_design_scores_gemma":[0.000044603945,0.00011037235,0.0012125934,0.00003842956,0.000112274596,0.000107288884,0.00014638873,0.9312439,0.020282906,0.04036074,0.006301579,0.00003882279],"about_ca_topic_score_codex":0.0062461197,"about_ca_topic_score_gemma":0.006248933,"teacher_disagreement_score":0.0062461197,"about_ca_system_score_codex":0.0012341846,"about_ca_system_score_gemma":0.0014211114,"threshold_uncertainty_score":0.012419522},"labels":[],"label_agreement":null},{"id":"W2151752770","doi":"10.1023/b:inrt.0000011209.19643.e2","title":"Augmenting Naive Bayes Classifiers with Statistical Language Models","year":2004,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Topic Modeling","field":"Computer Science","cited_by":247,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Naive Bayes classifier; Artificial intelligence; Computer science; Machine learning; Bayes error rate; Bayes classifier; Bayes' theorem; Classifier (UML); Conditional independence; Bayesian programming; Natural language processing; Pattern recognition (psychology); Bayes factor; Bayesian probability; Support vector machine","score_opus":0.013246135569706644,"score_gpt":0.23719751641258194,"score_spread":0.2239513808428753,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2151752770","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022382993,0.0048596133,0.95856935,0.0012830836,0.00078910723,0.0002622874,0.0008154029,0.006360177,0.00467802],"genre_scores_gemma":[0.30742505,0.0035352076,0.6714925,0.0011902794,0.0017439966,0.0005941053,0.0047144527,0.00080938835,0.0084950635],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99401516,0.00296964,0.00042591227,0.00075318996,0.0015950232,0.0002410549],"domain_scores_gemma":[0.9773309,0.017233443,0.00047390812,0.0014046349,0.0033459791,0.00021105814],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0077082305,0.0019560058,0.0028207814,0.0044482276,0.0011497216,0.0031331896,0.0023705685,0.0024643699,0.0049339873],"category_scores_gemma":[0.031472478,0.0010347067,0.0020054833,0.0036352542,0.00061085395,0.007949124,0.001710059,0.0031287754,0.007152606],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058743753,0.0006005458,0.0030491687,0.0005393695,0.00047378088,0.00014655326,0.00020473482,0.06320536,0.0057276585,0.006089303,0.02235456,0.89702153],"study_design_scores_gemma":[0.000088580775,0.00013295432,0.00073575904,0.00008684809,0.00032677408,0.00017731225,0.00007593527,0.9551525,0.003909549,0.032687556,0.0065668044,0.000059372174],"about_ca_topic_score_codex":0.0072441846,"about_ca_topic_score_gemma":0.010279382,"teacher_disagreement_score":0.0077082305,"about_ca_system_score_codex":0.00087294384,"about_ca_system_score_gemma":0.0016383493,"threshold_uncertainty_score":0.040765524},"labels":[],"label_agreement":null},{"id":"W2556771250","doi":"10.1007/s10791-016-9290-6","title":"Efficient distributed selective search","year":2016,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Directorate for Computer and Information Science and Engineering; Australian Research Council; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Ranking (information retrieval); Resource (disambiguation); Index (typography); Mirroring; Data mining; Selection (genetic algorithm); Web search query; Query expansion; Search engine; Information retrieval; Distributed computing; Machine learning; Computer network; World Wide Web","score_opus":0.013532769585453928,"score_gpt":0.25299165861495426,"score_spread":0.23945888902950033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2556771250","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15543671,0.0013623459,0.80657965,0.0007676325,0.00015809327,0.00024280777,0.0006189715,0.004950657,0.029883236],"genre_scores_gemma":[0.83824533,0.00035475817,0.14083803,0.00013994842,0.00014052539,0.00018110672,0.0006968207,0.00025062254,0.019152837],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99878806,0.00027392572,0.000049458584,0.00020476467,0.00047682642,0.00020698739],"domain_scores_gemma":[0.9970944,0.0010588433,0.00012760061,0.001100201,0.00049263315,0.00012625937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009818925,0.0006345296,0.0013924469,0.0011459755,0.00096275087,0.0014007444,0.0017200526,0.0008989691,0.009528766],"category_scores_gemma":[0.004595172,0.000353717,0.0004040439,0.001752265,0.00063717033,0.0023535194,0.001781361,0.00063265703,0.0029030142],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024943026,0.001006585,0.0034730572,0.0003973843,0.00014327475,0.00026432046,0.00026434517,0.15079038,0.045559227,0.07728012,0.03810048,0.6802265],"study_design_scores_gemma":[0.00013605852,0.00010560686,0.0007408044,0.000007698695,0.000050705723,0.00018608668,0.000070378526,0.9465881,0.007496448,0.0404508,0.004153468,0.000013721636],"about_ca_topic_score_codex":0.002521502,"about_ca_topic_score_gemma":0.004615814,"teacher_disagreement_score":0.009528766,"about_ca_system_score_codex":0.0009296191,"about_ca_system_score_gemma":0.0019592973,"threshold_uncertainty_score":0.031876862},"labels":[],"label_agreement":null},{"id":"W2576646839","doi":"10.1007/s10791-016-9292-4","title":"Enhancing click models with mouse movement information","year":2017,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Relevance (law); Computer science; Information retrieval; Search engine; Process (computing); Movement (music); Artificial intelligence; World Wide Web; Human–computer interaction; Programming language","score_opus":0.01890283138193856,"score_gpt":0.25183785941004816,"score_spread":0.23293502802810961,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2576646839","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29236707,0.0009142594,0.689815,0.000570634,0.00015953521,0.00014791651,0.00043514644,0.00666689,0.008923553],"genre_scores_gemma":[0.9427782,0.00036078252,0.05044261,0.00011283425,0.00004123841,0.000079680525,0.00030812016,0.00038859085,0.0054878625],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99950826,0.0001708075,0.000024822557,0.000078297344,0.00016338608,0.00005449936],"domain_scores_gemma":[0.9963174,0.0024281635,0.00024700697,0.0003950546,0.0004395333,0.00017273675],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00085592596,0.0010221355,0.0007125706,0.00077625754,0.00019948419,0.0011125566,0.0008591985,0.0011903379,0.005631525],"category_scores_gemma":[0.008251334,0.00031151768,0.000524937,0.0006059538,0.00028049323,0.002572003,0.0007720094,0.0007977426,0.0016639694],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039278213,0.002944244,0.011364824,0.0006284298,0.00018925374,0.0002996876,0.0003047657,0.40801913,0.19167472,0.020747393,0.009900971,0.34999883],"study_design_scores_gemma":[0.000053563163,0.00027403462,0.0009497042,0.000014252044,0.000047673057,0.00005467052,0.000016100026,0.98337907,0.011480627,0.0028816306,0.00082816585,0.00002060089],"about_ca_topic_score_codex":0.0031233486,"about_ca_topic_score_gemma":0.004271904,"teacher_disagreement_score":0.005631525,"about_ca_system_score_codex":0.00057048316,"about_ca_system_score_gemma":0.0006215954,"threshold_uncertainty_score":0.01883936},"labels":[],"label_agreement":null},{"id":"W2584295907","doi":"10.1007/s10791-017-9294-x","title":"Constructing click models for search users","year":2017,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Image Retrieval and Classification Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; World Wide Web; Information retrieval","score_opus":0.05359408056046548,"score_gpt":0.31268713016859917,"score_spread":0.2590930496081337,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2584295907","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36320934,0.000887703,0.61808497,0.0020991408,0.00014855208,0.00038017108,0.0050748517,0.0038505672,0.006264576],"genre_scores_gemma":[0.9502027,0.00051164196,0.034843344,0.00017374544,0.00016187478,0.00035274556,0.004239736,0.000343134,0.009171203],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983981,0.00060340326,0.00008284069,0.0003205653,0.00026525333,0.00032975595],"domain_scores_gemma":[0.98701954,0.00967071,0.00057377556,0.0011779746,0.0010561789,0.0005018943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040578707,0.001397154,0.0017348643,0.00457175,0.0007562934,0.0030586917,0.0026572265,0.0037243953,0.006803075],"category_scores_gemma":[0.019441925,0.0011491205,0.002714127,0.0035682267,0.0012146437,0.0059833294,0.0014040619,0.0028522424,0.0029645436],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032471747,0.0020530694,0.08730829,0.00052710145,0.000658115,0.0007000459,0.0011733522,0.54355377,0.0076481868,0.116407,0.026597176,0.21012667],"study_design_scores_gemma":[0.000019053135,0.00004204756,0.001407067,0.000013439976,0.000044058237,0.000040240084,0.00004048965,0.986545,0.0005166799,0.010964438,0.0003514713,0.000016101676],"about_ca_topic_score_codex":0.02195994,"about_ca_topic_score_gemma":0.023358945,"teacher_disagreement_score":0.02195994,"about_ca_system_score_codex":0.0024790897,"about_ca_system_score_gemma":0.0017777982,"threshold_uncertainty_score":0.043664217},"labels":[],"label_agreement":null},{"id":"W2803782109","doi":"10.1007/s10791-018-9332-3","title":"(CF)2 architecture: contextual collaborative filtering","year":2018,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Oakville-Trafalgar Memorial Hospital; Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Collaborative filtering; Computer science; Recommender system; Context (archaeology); Architecture; The Internet; Selection (genetic algorithm); Fraction (chemistry); Machine learning; Scale (ratio); Artificial intelligence; Contextual design; Filter (signal processing); Information retrieval; Data science; World Wide Web; Human–computer interaction","score_opus":0.011036141244884471,"score_gpt":0.24795552541158117,"score_spread":0.2369193841666967,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2803782109","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016543323,0.00052902213,0.95861644,0.00048648624,0.00018907596,0.00031614958,0.00076597015,0.015428868,0.007124633],"genre_scores_gemma":[0.36361495,0.00041296476,0.6181422,0.00061358995,0.00023067661,0.00052422803,0.0020435294,0.00029259856,0.014125233],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992668,0.00012955986,0.000050493185,0.00030053928,0.00014485627,0.00010771232],"domain_scores_gemma":[0.9989371,0.00019039489,0.000046850942,0.00040464083,0.00034493292,0.00007606196],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012372702,0.00081225706,0.0009036288,0.0011229524,0.00088954583,0.0012869113,0.0028903093,0.0020619722,0.009769381],"category_scores_gemma":[0.00328225,0.0004207328,0.0009970374,0.0012255878,0.0005210144,0.0019703207,0.001452053,0.0010824152,0.004081005],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008329091,0.0005125988,0.008012363,0.00029393626,0.00031832382,0.00028707305,0.0003367763,0.1162094,0.016472613,0.030644193,0.04630604,0.77977383],"study_design_scores_gemma":[0.000070774055,0.00025430266,0.0017751646,0.000034901936,0.00011427076,0.0003189746,0.000051369447,0.94685125,0.009331486,0.013076487,0.028039701,0.00008118883],"about_ca_topic_score_codex":0.028478358,"about_ca_topic_score_gemma":0.03935464,"teacher_disagreement_score":0.028478358,"about_ca_system_score_codex":0.00079067744,"about_ca_system_score_gemma":0.0016044894,"threshold_uncertainty_score":0.056625187},"labels":[],"label_agreement":null},{"id":"W2850470857","doi":"10.1007/s10791-018-9337-y","title":"User interest prediction over future unobserved topics on social networks","year":2018,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Focus (optics); Work (physics); User modeling; Data science; Social network (sociolinguistics); World Wide Web; Social media; User interface; Engineering","score_opus":0.02993305481423278,"score_gpt":0.2657125739616944,"score_spread":0.23577951914746162,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2850470857","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.63854307,0.0023837772,0.35105166,0.0013966206,0.00012801337,0.000116935604,0.002349428,0.00057032786,0.0034601605],"genre_scores_gemma":[0.9807387,0.00063312874,0.014616,0.000050735987,0.00018125055,0.000036614405,0.0015624298,0.0000199126,0.002161236],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993617,0.00021008584,0.000042017487,0.0001882621,0.00013637303,0.00006155601],"domain_scores_gemma":[0.9951997,0.003594409,0.0003026162,0.00034710357,0.00042535705,0.00013079867],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013383306,0.0006047754,0.000764186,0.0022618973,0.00034457864,0.0010981425,0.0008394047,0.0009796218,0.0011536694],"category_scores_gemma":[0.005838231,0.0003540031,0.0007518898,0.0016456908,0.0002711148,0.002414616,0.0005002736,0.0011505958,0.00078579056],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021377332,0.0015519744,0.27728477,0.000577836,0.0010915766,0.0006389449,0.0009321478,0.30375013,0.020474765,0.0134830335,0.011399619,0.3666774],"study_design_scores_gemma":[0.000014891912,0.000094701776,0.014129451,0.000014727823,0.000111285844,0.00009010141,0.000059874004,0.9793241,0.0015987199,0.003781282,0.0007675905,0.000013320563],"about_ca_topic_score_codex":0.005740302,"about_ca_topic_score_gemma":0.0108568175,"teacher_disagreement_score":0.005740302,"about_ca_system_score_codex":0.00053158036,"about_ca_system_score_gemma":0.00031488526,"threshold_uncertainty_score":0.011413813},"labels":[],"label_agreement":null},{"id":"W2967813257","doi":"10.1007/s10791-019-09361-0","title":"Evaluating sentence-level relevance feedback for high-recall information retrieval","year":2019,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Topic Modeling","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Relevance feedback; Relevance (law); Recall; Computer science; Sentence; Baseline (sea); Information retrieval; Precision and recall; Natural language processing; Artificial intelligence; Cognitive psychology; Psychology; Image retrieval","score_opus":0.046889037076321534,"score_gpt":0.292025163440755,"score_spread":0.2451361263644335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2967813257","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.75100553,0.015410303,0.21872827,0.0008884288,0.0008114804,0.0008568126,0.0013804425,0.0057659834,0.005152695],"genre_scores_gemma":[0.9170604,0.0011110565,0.074843355,0.00022563324,0.00047239405,0.00019080921,0.0027117739,0.00019716885,0.0031874273],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99647385,0.0017783897,0.00027790756,0.00037794345,0.0009265538,0.00016531562],"domain_scores_gemma":[0.9842606,0.012574472,0.00046355146,0.0003900566,0.0020153825,0.00029586742],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059652943,0.0011963488,0.0015838961,0.001913242,0.00059730833,0.0012882733,0.0008333756,0.0018683671,0.0033165875],"category_scores_gemma":[0.02758873,0.0003225321,0.00069064717,0.00072589726,0.0003188745,0.0015665174,0.0007450907,0.0008624804,0.0015494417],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.010543714,0.002388965,0.009490937,0.0026289432,0.00092818943,0.00043279587,0.0005421565,0.044052932,0.14398175,0.0010948478,0.01943057,0.7644842],"study_design_scores_gemma":[0.00087746,0.004431417,0.018350787,0.00011072237,0.0012011834,0.00056732364,0.00024981386,0.9054238,0.063661344,0.00205599,0.00294272,0.00012732879],"about_ca_topic_score_codex":0.0037368739,"about_ca_topic_score_gemma":0.0049501276,"teacher_disagreement_score":0.0059652943,"about_ca_system_score_codex":0.00070017274,"about_ca_system_score_gemma":0.0011364227,"threshold_uncertainty_score":0.031547844},"labels":[],"label_agreement":null},{"id":"W2976855657","doi":"10.1007/s10791-019-09364-x","title":"ReBoost: a retrieval-boosted sequence-to-sequence model for neural response generation","year":2019,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Topic Modeling","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Computer science; Benchmark (surveying); Conversation; Sequence (biology); Process (computing); Artificial intelligence; Natural language generation; Artificial neural network; Language model; Natural language processing; Recurrent neural network; Machine learning; Natural language; Programming language; Linguistics","score_opus":0.06926171490965688,"score_gpt":0.29959857846808324,"score_spread":0.23033686355842636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2976855657","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0124847805,0.0016706721,0.9764766,0.00040270877,0.00027719451,0.00018415894,0.00047849747,0.007150414,0.000874906],"genre_scores_gemma":[0.3098129,0.0012530565,0.6651555,0.0011058195,0.0005669449,0.00093753287,0.0032643736,0.0012766805,0.016627235],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989858,0.00036090385,0.00005319845,0.00025191015,0.00020676399,0.00014137538],"domain_scores_gemma":[0.9986563,0.00070633215,0.00007757792,0.000123855,0.00035486696,0.00008095574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023317148,0.0015646098,0.002680978,0.001497456,0.00065042154,0.0009734162,0.0042623007,0.0029234914,0.004797721],"category_scores_gemma":[0.0041233106,0.0007834593,0.0014742132,0.0017304125,0.00058068056,0.0016675759,0.0013692753,0.0033647686,0.0029566728],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009091464,0.0006341359,0.0007158202,0.00036339313,0.00033575587,0.00011762115,0.00008754877,0.28351018,0.012132262,0.005943159,0.02459648,0.67065454],"study_design_scores_gemma":[0.000021823107,0.000052857347,0.00008452019,0.000008001718,0.000017480377,0.000018715096,0.000004084248,0.9949279,0.0017742163,0.00224053,0.0008398411,0.000010054155],"about_ca_topic_score_codex":0.011684983,"about_ca_topic_score_gemma":0.013936229,"teacher_disagreement_score":0.011684983,"about_ca_system_score_codex":0.0010987844,"about_ca_system_score_gemma":0.0016416564,"threshold_uncertainty_score":0.02323395},"labels":[],"label_agreement":null},{"id":"W3044191208","doi":"10.1007/s10791-020-09379-9","title":"Robust keyword search in large attributed graphs","year":2020,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Complex Network Analysis Techniques","field":"Physics and Astronomy","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); Toronto Metropolitan University; University of Waterloo; Ontario Tech University","funders":"","keywords":"Scalability; Computer science; SPARK (programming language); Theoretical computer science; Set (abstract data type); Graph; Keyword search; Data mining; Information retrieval; Database","score_opus":0.0316519411098354,"score_gpt":0.26000995574858576,"score_spread":0.22835801463875036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3044191208","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06803502,0.0018734331,0.9231253,0.0008518716,0.00006954399,0.00016029712,0.0021298132,0.0020337729,0.0017209684],"genre_scores_gemma":[0.73592126,0.0018392361,0.24795502,0.00029801336,0.00028653053,0.00017761368,0.005533916,0.00045973624,0.0075287344],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972365,0.0008453807,0.00022656973,0.0006501672,0.0007953383,0.00024607283],"domain_scores_gemma":[0.9788225,0.01590914,0.001738935,0.0020507334,0.0011082307,0.00037038256],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027634508,0.0010738841,0.0025225384,0.006952778,0.0011129391,0.0027996474,0.0027502691,0.0016379171,0.0024889174],"category_scores_gemma":[0.02370071,0.0010587359,0.001241565,0.00818918,0.001159131,0.0065806694,0.0022413458,0.0014551877,0.0010745595],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007120758,0.00046025307,0.0049116304,0.0010836448,0.000355848,0.0004036022,0.00031761444,0.66459906,0.009484603,0.046196323,0.012949669,0.2585257],"study_design_scores_gemma":[0.00001898809,0.000023320183,0.00035831556,0.000013472022,0.000038509617,0.000050024475,0.00005216168,0.9507022,0.0014920513,0.046422567,0.00081772014,0.000010701162],"about_ca_topic_score_codex":0.012695221,"about_ca_topic_score_gemma":0.014777907,"teacher_disagreement_score":0.012695221,"about_ca_system_score_codex":0.0019812824,"about_ca_system_score_gemma":0.0019665156,"threshold_uncertainty_score":0.025242686},"labels":[],"label_agreement":null},{"id":"W3133365413","doi":"10.1007/s10791-021-09398-0","title":"Neural ranking models for document retrieval","year":2021,"lang":"en","type":"preprint","venue":"Information Retrieval","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Instituto de Ciencias del Mar y Limnología, Universidad Nacional Autónoma de México; National Institute of Standards and Technology; Institute for Catastrophic Loss Reduction; National Science Foundation","keywords":"Ranking (information retrieval); Computer science; Artificial intelligence; Machine learning; Set (abstract data type); Variety (cybernetics); Information retrieval; Artificial neural network; Deep learning; Learning to rank; Data mining","score_opus":0.02256006568427007,"score_gpt":0.29165840351195466,"score_spread":0.2690983378276846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3133365413","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.105906874,0.012281385,0.8390996,0.0043254434,0.000546144,0.00025319573,0.0015675529,0.00343275,0.03258703],"genre_scores_gemma":[0.8747833,0.0035065752,0.07991028,0.00059684133,0.0005817625,0.000323511,0.0014830643,0.00026308987,0.03855155],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992176,0.00029710119,0.000052529467,0.00014213286,0.00017828816,0.00011232211],"domain_scores_gemma":[0.9978795,0.0011181682,0.00024303254,0.0001667478,0.00050747395,0.00008515419],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020558096,0.00080945,0.0013602833,0.0019137101,0.00044864154,0.002023888,0.0016125607,0.0015463745,0.008365574],"category_scores_gemma":[0.006747213,0.00037738887,0.0008614809,0.0020979694,0.00065180194,0.0027603407,0.00067140226,0.0014695016,0.0027884739],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002744056,0.0002551485,0.0017188299,0.0003829457,0.00015284163,0.00014547426,0.0001144791,0.71367806,0.002390051,0.10800232,0.018563302,0.15432219],"study_design_scores_gemma":[0.000010149549,0.000021334905,0.00019924996,0.000011065277,0.000011085252,0.000017407894,0.000006081404,0.9778211,0.00015539832,0.020919751,0.0008187542,0.0000086821465],"about_ca_topic_score_codex":0.0074381493,"about_ca_topic_score_gemma":0.007762405,"teacher_disagreement_score":0.008365574,"about_ca_system_score_codex":0.0019865532,"about_ca_system_score_gemma":0.0007170533,"threshold_uncertainty_score":0.027985632},"labels":[],"label_agreement":null},{"id":"W3196754070","doi":"10.1007/s10791-022-09411-0","title":"Shallow pooling for sparse labels","year":2022,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Ranking (information retrieval); Mean reciprocal rank; Information retrieval; Pooling; Computer science; Relevance (law); Set (abstract data type); Rank (graph theory); Learning to rank; Preference; Artificial intelligence; Statistics; Mathematics; Combinatorics","score_opus":0.023324810570541036,"score_gpt":0.2593679142488165,"score_spread":0.23604310367827547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3196754070","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018839287,0.0008545947,0.97318596,0.00069981057,0.00009826892,0.000074271986,0.00080813735,0.0027889712,0.0026506635],"genre_scores_gemma":[0.59758407,0.0012037638,0.3711297,0.00075018255,0.00043748767,0.00034371892,0.005846831,0.0008018462,0.021902438],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99922085,0.00018402701,0.000048191498,0.00022702142,0.00015250075,0.00016738617],"domain_scores_gemma":[0.99806005,0.0009865154,0.0001309907,0.0005402643,0.00017973696,0.00010241084],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001719401,0.0014325066,0.0021215656,0.0011667408,0.00074618537,0.0018151022,0.0021246648,0.0020585742,0.008960826],"category_scores_gemma":[0.006325822,0.0008750011,0.0013816798,0.0015767127,0.0011295613,0.0046860483,0.0029858297,0.0023147913,0.0022753812],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007602893,0.0003662741,0.0013644757,0.0005191336,0.00026265313,0.00023626928,0.00025545704,0.12903613,0.021770693,0.10748446,0.032456875,0.7054873],"study_design_scores_gemma":[0.000028833525,0.00006922768,0.00040702932,0.000031247404,0.000057960726,0.000041782394,0.000026364572,0.8792862,0.0043807025,0.1134881,0.0021602183,0.000022191161],"about_ca_topic_score_codex":0.009584728,"about_ca_topic_score_gemma":0.0147181135,"teacher_disagreement_score":0.009584728,"about_ca_system_score_codex":0.0014557124,"about_ca_system_score_gemma":0.0015628084,"threshold_uncertainty_score":0.029976904},"labels":[],"label_agreement":null},{"id":"W4387453959","doi":"10.1007/s10791-023-09421-6","title":"Learning heterogeneous subgraph representations for team discovery","year":2023,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Advanced Graph Neural Networks","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph; York University; Toronto Metropolitan University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Overfitting; Ranking (information retrieval); Machine learning; Set (abstract data type); Graph; Task (project management); Artificial intelligence; Baseline (sea); Representation (politics); Data science; Artificial neural network; Theoretical computer science","score_opus":0.014467129190787973,"score_gpt":0.27384187858459824,"score_spread":0.2593747493938103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387453959","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10244079,0.0015133278,0.8831118,0.0012165336,0.00015656983,0.00023490132,0.0033524428,0.0032908993,0.0046827868],"genre_scores_gemma":[0.7187168,0.0008339609,0.2631214,0.00041381465,0.00021127178,0.00028909888,0.010562416,0.00036810944,0.005483144],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991055,0.00025979403,0.00004908142,0.00030650976,0.00015916523,0.00012001069],"domain_scores_gemma":[0.9977761,0.0010689587,0.000244702,0.00050815695,0.00023007914,0.00017210798],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000920961,0.0009869824,0.0013699658,0.004785508,0.0010308238,0.001564969,0.0022093283,0.002163976,0.0037126353],"category_scores_gemma":[0.0066466294,0.0005319881,0.0015028163,0.005129785,0.00076655275,0.003887212,0.001990564,0.0018160725,0.0013117809],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008524562,0.0010549609,0.010477688,0.0006929187,0.00064124155,0.00044045836,0.0006172446,0.27717432,0.007795522,0.06788401,0.05477018,0.577599],"study_design_scores_gemma":[0.00004078356,0.00005533857,0.00061156496,0.000040661496,0.00008859437,0.00006318796,0.00013065558,0.8740997,0.0011633583,0.12105311,0.0026378534,0.000015189688],"about_ca_topic_score_codex":0.007990863,"about_ca_topic_score_gemma":0.016301813,"teacher_disagreement_score":0.007990863,"about_ca_system_score_codex":0.0012678348,"about_ca_system_score_gemma":0.0013389668,"threshold_uncertainty_score":0.015888691},"labels":[],"label_agreement":null}]}