{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":630,"total_is_capped":false,"direct_labels_cover":3,"predictions_cover":630,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"371cb4d87b7b","filters":{"topic":"Web Data Mining and Analysis"}},"results":[{"id":"W2913556077","doi":"","title":"Proceedings of the 25th International Conference on World Wide Web","year":2016,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":635,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université du Québec à Montréal; Université TÉLUQ","funders":"","keywords":"Computer science; World Wide Web; Track (disk drive); Social media; Consistency (knowledge bases); Personalization; Phishing; Data science; The Internet","authors":[{"name":"Jacqueline Bourdeau","is_ca":true},{"name":"James Hendler","is_ca":false},{"name":"Roger Nkambou","is_ca":true},{"name":"Ian Horrocks","is_ca":false},{"name":"Ben Y. Zhao","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0271719921717674,"gpt":0.2464222719911995,"spread":0.2192502798194321,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003263557,0.001665303,0.002333171,0.002737648,0.001465721,0.009391792,0.002019842,0.002376351,0.1468257],"category_scores_gemma":[0.008963847,0.0005060015,0.001051105,0.002789654,0.001136092,0.01021781,0.003375754,0.003768871,0.1359949],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009459822,"about_ca_system_score_gemma":0.002246858,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001824705,"about_ca_topic_score_gemma":0.002499146,"domain_scores_codex":[0.9965071,0.0008328085,0.000334178,0.0006074866,0.001365594,0.0003527438],"domain_scores_gemma":[0.9933648,0.001370663,0.0003167014,0.001275816,0.00241857,0.001253553],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008432579,0.0001076978,0.0006485438,0.0005566505,0.00006197642,0.0001700268,0.0001201776,0.0001257967,0.001833711,0.004439402,0.7940987,0.197753],"study_design_scores_gemma":[0.000008182868,0.00003366674,0.0008641682,0.0002261705,0.00002600145,0.0002354255,0.0001332308,0.0006980333,0.0004923221,0.003542114,0.9937196,0.00002096649],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.01498295,0.1494487,0.1073816,0.0452598,0.1682955,0.001686197,0.01974892,0.01765665,0.4755397],"genre_scores_gemma":[0.03963891,0.07584099,0.04264375,0.01138801,0.025027,0.0009412225,0.04931273,0.004014378,0.7511929],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.1468257,"threshold_uncertainty_score":0.4911811,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2090018148","doi":"10.1109/dnsr.2004.1344743","title":"Weighted PageRank algorithm","year":2004,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":562,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of New Brunswick","funders":"Atlantic Canada Opportunities Agency","keywords":"PageRank; Computer science; Ranking (information retrieval); HITS algorithm; Web page; Information retrieval; Popularity; Rank (graph theory); Categorization; Learning to rank; Data mining; World Wide Web; Theoretical computer science; Web search engine; Web navigation; Artificial intelligence; Mathematics","authors":[{"name":"W. Xing","is_ca":true},{"name":"Ali A. Ghorbani","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.008709172515255128,"gpt":0.218913923281581,"spread":0.2102047507663259,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002237172,0.002159057,0.003130384,0.004688956,0.001447015,0.003624204,0.003137503,0.002753312,0.02266563],"category_scores_gemma":[0.009484827,0.0007167762,0.0009801054,0.00542103,0.0008624669,0.00435491,0.001353948,0.001402621,0.02436554],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007499558,"about_ca_system_score_gemma":0.002355318,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002438524,"about_ca_topic_score_gemma":0.003572599,"domain_scores_codex":[0.9960726,0.001334611,0.00033036,0.000566407,0.001405591,0.000290505],"domain_scores_gemma":[0.9961535,0.001456131,0.0002956435,0.0006252232,0.001328528,0.000140884],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003266118,0.0002957711,0.00136553,0.0006956443,0.0002661227,0.0002466542,0.00008140763,0.10949,0.002896812,0.0387013,0.09781807,0.7478162],"study_design_scores_gemma":[0.0003808018,0.000340955,0.0005573682,0.0001390659,0.0001553787,0.001002291,0.000114235,0.7485837,0.004785503,0.1226579,0.12119,0.00009274607],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005974591,0.003628164,0.9565156,0.0007866667,0.000961598,0.0008792797,0.001500049,0.005222431,0.02453162],"genre_scores_gemma":[0.1269929,0.004750919,0.8093464,0.0006439817,0.001261041,0.001148127,0.006622575,0.0007233185,0.04851066],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02266563,"threshold_uncertainty_score":0.07582414,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2137763598","doi":"10.1109/tkde.2004.58","title":"Efficient phrase-based document indexing for Web document clustering","year":2004,"lang":"en","type":"article","venue":"IEEE Transactions on Knowledge and Data Engineering","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":320,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Document clustering; Computer science; Cluster analysis; Vector space model; Search engine indexing; Phrase; Information retrieval; Document classification; Data mining; tf–idf; Artificial intelligence; Term (time)","authors":[{"name":"Khaled M. Hammouda","is_ca":true},{"name":"Mohamed S. Kamel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02028388213544066,"gpt":0.2679597706680517,"spread":0.247675888532611,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001256536,0.000784139,0.001399069,0.004405822,0.001098526,0.00160535,0.001782839,0.001007364,0.002336452],"category_scores_gemma":[0.007110163,0.0004591109,0.001024594,0.009499787,0.0006611206,0.002796623,0.001643425,0.001270631,0.002618118],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001103874,"about_ca_system_score_gemma":0.001569829,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004000303,"about_ca_topic_score_gemma":0.004178685,"domain_scores_codex":[0.9983574,0.0004335535,0.0001242489,0.0002648279,0.0007371228,0.00008283414],"domain_scores_gemma":[0.9980203,0.0006195031,0.0001676637,0.0005740468,0.0005630313,0.00005553494],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002013628,0.0001207917,0.0008992539,0.0004222789,0.0001033649,0.0001075881,0.0002800754,0.0887168,0.02255181,0.05916213,0.01699824,0.8104363],"study_design_scores_gemma":[0.00004592617,0.00008310653,0.0005826834,0.00002482439,0.0000490034,0.0002336378,0.00007138492,0.9032925,0.008853515,0.07417627,0.01253762,0.00004959584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003313171,0.0003755327,0.993883,0.0000627565,0.00003433012,0.00009091949,0.0003639852,0.00116452,0.0007118771],"genre_scores_gemma":[0.03637255,0.0005158997,0.9601054,0.00004729038,0.00007315654,0.0002402213,0.001676647,0.00021147,0.0007574298],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004405822,"threshold_uncertainty_score":0.008009255,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2011539648","doi":"10.1145/2109205.2109208","title":"Crawling Ajax-Based Web Applications through Dynamic Analysis of User Interface State Changes","year":2012,"lang":"en","type":"article","venue":"ACM Transactions on the Web","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":298,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Ajax; Computer science; JavaScript; Dynamic web page; World Wide Web; Crawling; Web application; Web page; Interactivity; Web modeling; Web-based simulation; Mashup; User interface; Web crawler; Web API; Programming language","authors":[{"name":"Ali Mesbah","is_ca":true},{"name":"Arie van Deursen","is_ca":false},{"name":"Stefan Lenselink","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0267733111547076,"gpt":0.2868866886135497,"spread":0.2601133774588421,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001402716,0.0008842014,0.0006297424,0.003148579,0.0008947005,0.001872547,0.0009280408,0.0006601916,0.0004194359],"category_scores_gemma":[0.00974799,0.0005727963,0.0005123622,0.00160819,0.000730879,0.002214871,0.001277933,0.0009653281,0.00033765],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005151225,"about_ca_system_score_gemma":0.001119463,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003250972,"about_ca_topic_score_gemma":0.006082758,"domain_scores_codex":[0.9981695,0.0004107309,0.0001452141,0.0004352949,0.0007555289,0.00008377858],"domain_scores_gemma":[0.993494,0.003203808,0.0009461509,0.001219151,0.0009713123,0.0001656718],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000526754,0.000939603,0.07817148,0.0006565663,0.0002815205,0.000823028,0.0026653,0.05564681,0.1420386,0.009274337,0.003711622,0.7052644],"study_design_scores_gemma":[0.00004419236,0.0002305462,0.02848814,0.00005501628,0.0001194101,0.0007762473,0.0003601455,0.868786,0.08320983,0.01018771,0.007661154,0.00008163029],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.312548,0.0004903547,0.6609089,0.0002117247,0.0000299803,0.0003834539,0.0004066984,0.02290726,0.002113614],"genre_scores_gemma":[0.5519527,0.000319165,0.4442876,0.00006480751,0.0000243474,0.0002117924,0.001317,0.0005754922,0.001247074],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003250972,"threshold_uncertainty_score":0.007418394,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2010463775","doi":"10.1145/860435.860449","title":"Query type classification for web document retrieval","year":2003,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":251,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Information retrieval; Relevance (law); Task (project management); Web query classification; Query expansion; Relevance feedback; Scheme (mathematics); Web search query; Search engine; Artificial intelligence; Image retrieval","authors":[{"name":"In-Ho Kang","is_ca":true},{"name":"Gil-Chang Kim","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03861046735755231,"gpt":0.2879217894911348,"spread":0.2493113221335825,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004512313,0.0007191086,0.001672775,0.006254131,0.001030044,0.001647767,0.00119203,0.001152058,0.002158492],"category_scores_gemma":[0.01409368,0.0003203835,0.001107628,0.00437265,0.0005353484,0.002726723,0.0007230038,0.0009381598,0.002200413],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001233685,"about_ca_system_score_gemma":0.001325761,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005153408,"about_ca_topic_score_gemma":0.00404668,"domain_scores_codex":[0.9957668,0.001647126,0.0005285114,0.000477082,0.00132443,0.0002562075],"domain_scores_gemma":[0.9944211,0.002437307,0.0003552905,0.001131365,0.001494259,0.0001607039],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008675567,0.0002537264,0.007848485,0.0005376589,0.0001174097,0.0002041109,0.0002256773,0.01060907,0.02573626,0.008920014,0.01884626,0.9258336],"study_design_scores_gemma":[0.0002540266,0.0006360821,0.01479115,0.0001423776,0.0003304457,0.001668013,0.0004106216,0.8606097,0.04242958,0.04434629,0.03416542,0.0002163122],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0721662,0.005724638,0.9058373,0.0008125321,0.0003165707,0.00100486,0.002286298,0.008345926,0.003505606],"genre_scores_gemma":[0.3272801,0.001680276,0.6595873,0.0003251476,0.0004972585,0.0009417003,0.005250444,0.0003969035,0.004040933],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006254131,"threshold_uncertainty_score":0.02386367,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4245995216","doi":"10.1007/978-3-642-45135-5","title":"Recommendation Systems in Software Engineering","year":2014,"lang":"en","type":"book","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":212,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Calgary; McGill University","funders":"","keywords":"Software engineering; Computer science","authors":[{"name":"Martin P. Robillard","is_ca":true},{"name":"Walid Maalej","is_ca":false},{"name":"Robert J. Walker","is_ca":true},{"name":"Thomas Zimmermann","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01132452895672102,"gpt":0.2009287867464777,"spread":0.1896042577897566,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007285682,0.00113042,0.00118798,0.001809961,0.0005659245,0.002186663,0.001252588,0.001000826,0.01794387],"category_scores_gemma":[0.002747408,0.0005251589,0.0005961978,0.004284193,0.0007715285,0.004235351,0.0008306876,0.001897469,0.01589909],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001061056,"about_ca_system_score_gemma":0.000942964,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0038435,"about_ca_topic_score_gemma":0.005563485,"domain_scores_codex":[0.9992254,0.0001025251,0.00004013655,0.0001191224,0.0004714447,0.000041321],"domain_scores_gemma":[0.999143,0.0003806288,0.00003168453,0.0001658745,0.0002418194,0.00003694513],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002039039,0.00005727975,0.000187917,0.0003536635,0.00004520695,0.00003245975,0.00008445852,0.002663292,0.000912038,0.04333989,0.1779924,0.7743111],"study_design_scores_gemma":[0.00002463461,0.00009374178,0.001299703,0.0004816179,0.00006780656,0.0004676387,0.0001194057,0.03006753,0.001748461,0.1779038,0.7876735,0.00005220462],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005907493,0.2063984,0.4978111,0.009098535,0.007331758,0.0002382879,0.0008138796,0.004156639,0.2682439],"genre_scores_gemma":[0.03803144,0.103508,0.2207155,0.002163939,0.004649586,0.0001998258,0.001505908,0.0007769574,0.6284489],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01794387,"threshold_uncertainty_score":0.0600282,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2158943277","doi":"10.1109/tvcg.2008.175","title":"VisGets: Coordinated Visualizations for Web-based Information Exploration and Discovery","year":2008,"lang":"en","type":"article","venue":"IEEE Transactions on Visualization and Computer Graphics","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":175,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto; University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Information retrieval; Web navigation; World Wide Web; RSS; Filter (signal processing); Exploratory search; Web search query; Visualization; Web page; Data visualization; Dimension (graph theory); Search engine; Data mining","authors":[{"name":"Marian Dörk","is_ca":true},{"name":"Sheelagh Carpendale","is_ca":true},{"name":"Christopher Collins","is_ca":true},{"name":"Carey Williamson","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02577688320399458,"gpt":0.2634437113893258,"spread":0.2376668281853312,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001841293,0.001753188,0.001011388,0.003254798,0.0007793746,0.004649756,0.001357389,0.001352147,0.01098983],"category_scores_gemma":[0.009667491,0.0008052494,0.001126466,0.002450597,0.0007846782,0.004579508,0.005113581,0.001763828,0.001959249],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005191091,"about_ca_system_score_gemma":0.0007111297,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0023055,"about_ca_topic_score_gemma":0.003133239,"domain_scores_codex":[0.9990714,0.0003886021,0.00008223215,0.0001157505,0.0002601414,0.00008195922],"domain_scores_gemma":[0.9955692,0.002704052,0.000274887,0.0006305489,0.0004732133,0.0003482365],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003567677,0.0004537367,0.007858132,0.002688487,0.0003952981,0.001319859,0.009814763,0.02401432,0.0782311,0.08553024,0.1148472,0.6712792],"study_design_scores_gemma":[0.0006420211,0.0008178957,0.007681282,0.0009512277,0.000291597,0.001671787,0.003793053,0.3632571,0.06464694,0.1681755,0.3874883,0.000583266],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01998484,0.001299863,0.9083993,0.0008064612,0.0001716136,0.0003215592,0.003364645,0.05993368,0.005718047],"genre_scores_gemma":[0.1606007,0.001305641,0.8244753,0.0003140298,0.0001157954,0.0008017748,0.004078789,0.005015922,0.00329193],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01098983,"threshold_uncertainty_score":0.03676462,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2080132606","doi":"10.14778/1938545.1938547","title":"Automatic wrappers for large scale web extraction","year":2011,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":135,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Noise (video); Scale (ratio); Noisy data; Extraction (chemistry); Data extraction; Data mining; Information extraction; Training set; Artificial intelligence; Machine learning; Information retrieval; Pattern recognition (psychology)","authors":[{"name":"Nilesh Dalvi","is_ca":false},{"name":"Ravi Kumar","is_ca":false},{"name":"Mohamed A. Soliman","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02415002081639386,"gpt":0.2448207655012949,"spread":0.2206707446849011,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002602196,0.001668507,0.001613682,0.004270602,0.001297967,0.002185107,0.001884611,0.001476621,0.003021597],"category_scores_gemma":[0.01139046,0.001200313,0.001828758,0.004085403,0.0008858842,0.004344186,0.003520199,0.002041779,0.006799802],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004966936,"about_ca_system_score_gemma":0.001346164,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000926681,"about_ca_topic_score_gemma":0.001707474,"domain_scores_codex":[0.9972093,0.0006067707,0.0004008726,0.0007326978,0.0008799857,0.0001704273],"domain_scores_gemma":[0.992426,0.001960974,0.000627239,0.003818888,0.00102461,0.0001421712],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002291385,0.0003347528,0.005182769,0.000811688,0.0003316313,0.0008256119,0.0005330854,0.02544559,0.04993481,0.01702368,0.04627993,0.8530673],"study_design_scores_gemma":[0.00006642007,0.000129073,0.003284306,0.0002080914,0.0002187438,0.001336194,0.000190671,0.6150351,0.2012489,0.09832586,0.07982334,0.0001333824],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003154902,0.0001929454,0.9661502,0.00007452469,0.00003154033,0.00009006762,0.0007244122,0.02909285,0.000488524],"genre_scores_gemma":[0.04740198,0.0002877382,0.9415964,0.0001583595,0.00006681975,0.0001996986,0.005358897,0.003137903,0.00179219],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004270602,"threshold_uncertainty_score":0.01376188,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2036101886","doi":"10.5430/air.v2n1p44","title":"Exploiting web scraping in a collaborative filtering- based approach to web advertising","year":2012,"lang":"en","type":"article","venue":"Artificial Intelligence Research","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":132,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Computer science; World Wide Web; Web page; Scraper site; Copying; Web mining; Web development; Static web page; Web modeling; Web analytics; Web navigation; Information retrieval; Data Web; Web intelligence","authors":[{"name":"Eloisa Vargiu","is_ca":false},{"name":"Mirko Urru","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2115775441977723,"gpt":0.410260957608213,"spread":0.1986834134104407,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004107118,0.0008189813,0.001403587,0.003870361,0.00186177,0.003021067,0.001678249,0.00203931,0.001228508],"category_scores_gemma":[0.008080201,0.0006562243,0.001448448,0.003410746,0.001014588,0.002156233,0.001165856,0.001130376,0.0007827381],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009196628,"about_ca_system_score_gemma":0.001066492,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01105182,"about_ca_topic_score_gemma":0.01303954,"domain_scores_codex":[0.9963226,0.001187663,0.0003060323,0.0007165942,0.001249246,0.0002178116],"domain_scores_gemma":[0.9927911,0.003688662,0.0005472398,0.001390834,0.001341695,0.0002404893],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007397262,0.001446224,0.01036871,0.0004416389,0.0005957544,0.001224169,0.002428713,0.1158025,0.05623011,0.02530379,0.005564742,0.7798539],"study_design_scores_gemma":[0.00006089207,0.0002913091,0.004141542,0.00004074229,0.0002518052,0.00109633,0.0002595475,0.9397591,0.02396388,0.01537472,0.01459874,0.000161483],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04540999,0.000579598,0.9452137,0.0003739807,0.00007204533,0.0002872158,0.00006570897,0.002481796,0.005516069],"genre_scores_gemma":[0.4049489,0.0003696035,0.5882555,0.0002444978,0.0001336901,0.0001174673,0.0001821129,0.000137716,0.005610571],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01105182,"threshold_uncertainty_score":0.02197492,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4400526908","doi":"10.1145/3626772.3657707","title":"Large Language Models can Accurately Predict Searcher Preferences","year":2024,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":132,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Microsoft (Canada)","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing","authors":[{"name":"Paul Thomas","is_ca":false},{"name":"Seth Spielman","is_ca":false},{"name":"Nick Craswell","is_ca":false},{"name":"Bhaskar Mitra","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07561307685075239,"gpt":0.3207369305742917,"spread":0.2451238537235393,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003934912,0.00122099,0.0008440518,0.001728744,0.0004904509,0.00191518,0.0007727705,0.001317441,0.003329099],"category_scores_gemma":[0.02667455,0.0005143131,0.0007492597,0.001332811,0.0004580048,0.004940157,0.00102574,0.002296686,0.004614863],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008836073,"about_ca_system_score_gemma":0.0007891445,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005657483,"about_ca_topic_score_gemma":0.01307887,"domain_scores_codex":[0.9978168,0.001094382,0.0001442722,0.0004569629,0.0003480228,0.0001394765],"domain_scores_gemma":[0.9803209,0.01579235,0.0008369018,0.001515101,0.001208756,0.0003260803],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00263177,0.0009692283,0.08887585,0.0008766433,0.0005902198,0.000403602,0.001548365,0.3749354,0.02694152,0.007963168,0.03530994,0.4589544],"study_design_scores_gemma":[0.00004752091,0.0001381541,0.00596815,0.00003904447,0.00004569588,0.0001073102,0.0001606759,0.9790316,0.003095741,0.009054905,0.00226655,0.00004465734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5408095,0.003284117,0.4238415,0.002813431,0.0002250518,0.0003269006,0.005663665,0.0104756,0.01256028],"genre_scores_gemma":[0.9461505,0.0003771155,0.04470861,0.0003633359,0.0001035013,0.0001645968,0.004041461,0.0004485297,0.003642329],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005657483,"threshold_uncertainty_score":0.02081001,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2129546202","doi":"10.1109/tkde.2009.174","title":"An Efficient Concept-Based Mining Model for Enhancing Text Clustering","year":2009,"lang":"en","type":"article","venue":"IEEE Transactions on Knowledge and Data Engineering","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":131,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Sentence; Natural language processing; Phrase; Term (time); Similarity (geometry); Semantics (computer science); Cluster analysis; Artificial intelligence; Document clustering; Meaning (existential); Word (group theory); Measure (data warehouse); Information retrieval; Similarity measure; Semantic similarity; Data mining; Linguistics","authors":[{"name":"Shady Shehata","is_ca":true},{"name":"Fakhri Karray","is_ca":true},{"name":"Mohamed S. Kamel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02779020595604768,"gpt":0.2810140212831417,"spread":0.253223815327094,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002401253,0.00120815,0.00156362,0.002691224,0.001174269,0.001540669,0.003165118,0.001481713,0.001851873],"category_scores_gemma":[0.005989402,0.000546024,0.0016378,0.003836383,0.0006500463,0.003774294,0.001281214,0.001601557,0.001382491],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00146753,"about_ca_system_score_gemma":0.002304716,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006205599,"about_ca_topic_score_gemma":0.006492814,"domain_scores_codex":[0.998032,0.0004218387,0.0001275843,0.0004775755,0.0008587477,0.00008221613],"domain_scores_gemma":[0.9980314,0.0007630727,0.0001266179,0.0001815688,0.0008494802,0.00004779686],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002577812,0.0004638395,0.002256577,0.0004817782,0.0002271726,0.0003603792,0.0006281056,0.3258717,0.01287088,0.05980281,0.01117904,0.5855999],"study_design_scores_gemma":[0.00001140514,0.00002872181,0.0001354591,0.00001142023,0.00001541556,0.000096718,0.00002659438,0.9864913,0.001324792,0.009396266,0.002448864,0.00001302773],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004177923,0.0001679203,0.9939487,0.00014104,0.00003156213,0.0001640004,0.0001361474,0.0005482856,0.0006844412],"genre_scores_gemma":[0.06240765,0.0003027297,0.9339467,0.0001623432,0.00004591308,0.0005506253,0.0007043881,0.00008416014,0.001795447],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006205599,"threshold_uncertainty_score":0.01269919,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2120376711","doi":"10.1145/634067.634291","title":"Integrating back, history and bookmarks in web browsers","year":2001,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":93,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Web browser; World Wide Web; Multimedia; The Internet","authors":[{"name":"Shaun Kaasten","is_ca":true},{"name":"Saul Greenberg","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01732150180113526,"gpt":0.215305818929065,"spread":0.1979843171279298,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005351156,0.0009734515,0.001282672,0.002552312,0.0008110934,0.005971628,0.002139273,0.001391316,0.006037438],"category_scores_gemma":[0.01967826,0.001471059,0.0008916397,0.00172619,0.001069828,0.01171282,0.003370174,0.001716697,0.004245549],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001089685,"about_ca_system_score_gemma":0.001720053,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01068295,"about_ca_topic_score_gemma":0.01318999,"domain_scores_codex":[0.9974473,0.0006041384,0.0003595341,0.0004292187,0.0009965947,0.0001632539],"domain_scores_gemma":[0.9855682,0.005065703,0.0007284647,0.00608152,0.001891799,0.0006643473],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00138761,0.0004283308,0.0194458,0.0007519362,0.0001606101,0.0008609089,0.004364089,0.01349068,0.01646091,0.05257311,0.05353647,0.8365396],"study_design_scores_gemma":[0.0002810258,0.0004854962,0.008141293,0.0006879765,0.0004856383,0.001785698,0.00115339,0.3274067,0.08929617,0.1003698,0.4690085,0.0008982571],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03195035,0.001131736,0.7619318,0.0009365286,0.0002608843,0.0003956358,0.001466489,0.18733,0.01459646],"genre_scores_gemma":[0.2814141,0.001388933,0.662307,0.00098933,0.0003495421,0.0003246419,0.002727152,0.01347528,0.03702411],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01068295,"threshold_uncertainty_score":0.02829993,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1828830618","doi":"","title":"World wide web site summarization","year":2004,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":92,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"","keywords":"Automatic summarization; Computer science; World Wide Web; Web site; Directory; Web navigation; Information retrieval; Task (project management); Web page; Web mapping; Data Web; Web modeling; Web standards; The Internet","authors":[{"name":"Yongzheng Zhang","is_ca":true},{"name":"A. Nur Zincir‐Heywood","is_ca":true},{"name":"Evangelos Milios","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.009892234255779498,"gpt":0.2168120506008852,"spread":0.2069198163451057,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001484092,0.0009207298,0.0007526575,0.006564312,0.0005799395,0.002623953,0.0008324292,0.0005764301,0.004087245],"category_scores_gemma":[0.009127116,0.0002205468,0.0004983972,0.005675755,0.0002125353,0.002177405,0.0009419884,0.0006788488,0.002896063],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005265724,"about_ca_system_score_gemma":0.0007545868,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002076552,"about_ca_topic_score_gemma":0.003130312,"domain_scores_codex":[0.9981949,0.0005352393,0.0002150326,0.0003159635,0.0006477744,0.00009107303],"domain_scores_gemma":[0.9949929,0.001032301,0.0006191017,0.001069459,0.00213959,0.0001464857],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002127532,0.0001412495,0.004773383,0.0009980671,0.0001393575,0.0001794949,0.0004442234,0.008117912,0.01625522,0.005163185,0.07026,0.8933151],"study_design_scores_gemma":[0.0001373171,0.0007706753,0.05122141,0.0004863083,0.0007860517,0.001561938,0.002865696,0.2325117,0.1077744,0.04314632,0.5584384,0.0002998125],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.107919,0.008690565,0.7651791,0.001624642,0.0007183994,0.002531097,0.04610176,0.0336793,0.03355608],"genre_scores_gemma":[0.3709119,0.003760863,0.4841532,0.0002554787,0.0004589366,0.0008785629,0.1195387,0.001365552,0.01867689],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006564312,"threshold_uncertainty_score":0.01367319,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2585438896","doi":"","title":"The data civilizer system","year":2017,"lang":"en","type":"article","venue":"Conference on Innovative Data Systems Research","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science","authors":[{"name":"Dong Deng","is_ca":false},{"name":"Raul Castro Fernandez","is_ca":false},{"name":"Ziawasch Abedjan","is_ca":false},{"name":"Sibo Wang","is_ca":false},{"name":"Michael Stonebraker","is_ca":false},{"name":"Ahmed K. Elmagarmid","is_ca":false},{"name":"Ihab F. Ilyas","is_ca":true},{"name":"Samuel Madden","is_ca":false},{"name":"Mourad Ouzzani","is_ca":false},{"name":"Nan Tang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.6414419922836844,"gpt":0.5044311967031525,"spread":0.1370107955805319,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00725413,0.001461874,0.001763607,0.006275731,0.001439812,0.006318755,0.002672391,0.001623552,0.1199788],"category_scores_gemma":[0.02564324,0.001515758,0.001438842,0.005629128,0.001162419,0.007126191,0.005836281,0.003323982,0.1106685],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001468501,"about_ca_system_score_gemma":0.004278447,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003654358,"about_ca_topic_score_gemma":0.002024312,"domain_scores_codex":[0.994228,0.001479811,0.0007081583,0.001422849,0.001715909,0.0004451861],"domain_scores_gemma":[0.9850706,0.002685582,0.000473312,0.008820027,0.002072307,0.0008781185],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001103628,0.0002016102,0.002651093,0.0003854377,0.0001651075,0.0002048712,0.0003120849,0.001514807,0.005294672,0.03471392,0.7811111,0.1723416],"study_design_scores_gemma":[0.0003511275,0.00008609988,0.001560797,0.00008932059,0.00006263683,0.0002184559,0.00009128368,0.01756111,0.01303822,0.01796874,0.9488799,0.00009229402],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"software","genre_gemma":"empirical","genre_scores_codex":[0.008063158,0.001111095,0.232193,0.004031873,0.001459352,0.001158295,0.1097401,0.571889,0.07035417],"genre_scores_gemma":[0.1158024,0.001648258,0.1754462,0.003348728,0.001157026,0.001850672,0.4921064,0.1012466,0.1073937],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1199788,"threshold_uncertainty_score":0.4013689,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1485672862","doi":"","title":"BlogScope: a system for online analysis of high volume text streams","year":2007,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Blogosphere; Computer science; Ranking (information retrieval); Relevance (law); Information retrieval; Volume (thermodynamics); Set (abstract data type); Data stream mining; World Wide Web; Data mining; Data science; The Internet","authors":[{"name":"Nilesh Bansal","is_ca":true},{"name":"Nick Koudas","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01347909899140815,"gpt":0.2599205676831217,"spread":0.2464414686917135,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001029062,0.001032317,0.000941913,0.003202661,0.0008235238,0.001891238,0.001413327,0.0008506305,0.01695322],"category_scores_gemma":[0.004277943,0.0005573308,0.000440701,0.002807742,0.0003789226,0.003472018,0.002038142,0.0009540956,0.009514514],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004645391,"about_ca_system_score_gemma":0.001038801,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0035498,"about_ca_topic_score_gemma":0.004672009,"domain_scores_codex":[0.9992334,0.00008480559,0.00007314717,0.0001993237,0.0003541133,0.00005526622],"domain_scores_gemma":[0.9977626,0.0008292273,0.0002183039,0.0004934141,0.0004533848,0.0002430898],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001783147,0.0003822853,0.01159127,0.001153229,0.0002472754,0.0008064393,0.000763068,0.002791018,0.03723643,0.004816331,0.5687476,0.369682],"study_design_scores_gemma":[0.0008618396,0.0004525144,0.02522549,0.0002243731,0.0002508499,0.001988581,0.0005338281,0.2912276,0.06734021,0.01702266,0.5943714,0.000500554],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"software","genre_gemma":"methods","genre_scores_codex":[0.02829076,0.001110973,0.2911152,0.0007195151,0.0005145404,0.001298961,0.06040754,0.6021429,0.01439962],"genre_scores_gemma":[0.207835,0.001515435,0.5889969,0.001176208,0.0006374908,0.002208338,0.1483754,0.01976252,0.02949277],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01695322,"threshold_uncertainty_score":0.05671418,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1563293148","doi":"10.1007/978-1-4471-0105-5_15","title":"How People Recognise Previously Seen Web Pages from Titles, URLs and Thumbnails","year":2002,"lang":"en","type":"book-chapter","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Calgary","funders":"","keywords":"Thumbnail; Computer science; World Wide Web; Information retrieval; Web page; String (physics); Image (mathematics); Artificial intelligence; Mathematics","authors":[{"name":"Shaun Kaasten","is_ca":true},{"name":"Saul Greenberg","is_ca":true},{"name":"Christopher Edwards","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02709610252049598,"gpt":0.1978387770905628,"spread":0.1707426745700668,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007857726,0.0006231501,0.00051315,0.001122203,0.000381995,0.003151535,0.0008107361,0.002066331,0.005759657],"category_scores_gemma":[0.005840003,0.0004057971,0.0006173692,0.0008415687,0.0004919241,0.007214469,0.0009929725,0.0009475928,0.007773699],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002211305,"about_ca_system_score_gemma":0.0002229068,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003929486,"about_ca_topic_score_gemma":0.003775195,"domain_scores_codex":[0.9995455,0.0000693118,0.00002299092,0.0001821284,0.0001365275,0.00004342181],"domain_scores_gemma":[0.998301,0.0009897638,0.00008498169,0.0001516177,0.0003664762,0.000106301],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0006356913,0.0002101797,0.02818078,0.0005785284,0.0002039237,0.0005015028,0.006229314,0.001264202,0.06193955,0.002123121,0.04507298,0.8530602],"study_design_scores_gemma":[0.000209287,0.001272957,0.2086558,0.0008960008,0.001311805,0.01262489,0.03640798,0.2643929,0.168773,0.0923178,0.2124299,0.0007078287],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6436484,0.008374853,0.2588032,0.004582754,0.001550685,0.0003324296,0.002845665,0.009006662,0.07085542],"genre_scores_gemma":[0.7523569,0.004785957,0.1604466,0.002359151,0.0003217299,0.0001982253,0.003856001,0.001116413,0.07455901],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005759657,"threshold_uncertainty_score":0.01926798,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1886218757","doi":"10.1109/icdm.2002.1183904","title":"Phrase-based document similarity based on an index graph model","year":2003,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Document clustering; Cluster analysis; Phrase; Vector space model; Information retrieval; Graph; Similarity (geometry); Matching (statistics); Set (abstract data type); Artificial intelligence; Data mining; Mathematics","authors":[{"name":"Khaled M. Hammouda","is_ca":true},{"name":"Mohamed S. Kamel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02070779077958386,"gpt":0.2583232720134422,"spread":0.2376154812338583,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001126039,0.0005454622,0.001229881,0.004967847,0.0005954102,0.001540435,0.001710898,0.001038477,0.001696005],"category_scores_gemma":[0.008384758,0.00025483,0.001070633,0.006765921,0.0008878278,0.004589661,0.00118135,0.0008153256,0.0008462379],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009275427,"about_ca_system_score_gemma":0.0008512974,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002752351,"about_ca_topic_score_gemma":0.003168185,"domain_scores_codex":[0.9979874,0.0005126411,0.0001446471,0.0003644417,0.0009225618,0.0000682548],"domain_scores_gemma":[0.9969439,0.00123474,0.0003234595,0.0006264049,0.0007758754,0.00009559886],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006243394,0.0004234713,0.007424873,0.0007343898,0.0004978825,0.0004106421,0.0007160804,0.2284798,0.04287295,0.1770761,0.01223889,0.5285006],"study_design_scores_gemma":[0.00003639422,0.0001691946,0.001683523,0.00002191839,0.0001011175,0.0004200691,0.00006840036,0.9312703,0.005752809,0.05632987,0.004080491,0.00006589514],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0318128,0.0003892923,0.9641405,0.0001515461,0.00005335997,0.0001462233,0.0004599362,0.000923099,0.001923199],"genre_scores_gemma":[0.4387927,0.0007965512,0.5551353,0.0001608413,0.0001930537,0.0004146661,0.001803828,0.0002665895,0.002436385],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004967847,"threshold_uncertainty_score":0.006729782,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2098181468","doi":"10.1109/hicss.2005.114","title":"Automatic Identification of Home Pages on the Web","year":2005,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":68,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"","keywords":"Identification (biology); Computer science; World Wide Web; Web page; Biology","authors":[{"name":"Alistair Kennedy","is_ca":true},{"name":"Michael Shepherd","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0194470054804127,"gpt":0.2397336240570869,"spread":0.2202866185766742,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004242469,0.0002948503,0.0004404297,0.005512543,0.0004527662,0.001357408,0.0003472884,0.0005282444,0.00202577],"category_scores_gemma":[0.002117557,0.0001882652,0.0002953387,0.001849285,0.0002250483,0.001206912,0.0005619791,0.000436025,0.001728307],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002482142,"about_ca_system_score_gemma":0.000292673,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002965024,"about_ca_topic_score_gemma":0.005142911,"domain_scores_codex":[0.9996141,0.00008116393,0.00002512544,0.00008516265,0.0001440765,0.0000503841],"domain_scores_gemma":[0.9984319,0.0004921959,0.0001927503,0.0002079458,0.0005756733,0.0000995697],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007962265,0.0004196741,0.1736936,0.0005891396,0.0001289054,0.001833204,0.001135307,0.003187215,0.04811512,0.003842104,0.02811001,0.7381495],"study_design_scores_gemma":[0.00006566491,0.0002505633,0.5271004,0.0002973563,0.0002396469,0.004556507,0.002757072,0.3221065,0.08394375,0.01169125,0.0468754,0.0001158631],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.8994268,0.002168907,0.06190924,0.0002346696,0.0001471911,0.0003533649,0.005381771,0.0061322,0.02424587],"genre_scores_gemma":[0.906306,0.0007988726,0.07426788,0.00007676402,0.00009063361,0.000087602,0.008886081,0.0002841151,0.009202112],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.005512543,"threshold_uncertainty_score":0.006776929,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2070347268","doi":"10.1145/1753326.1753426","title":"A study of tabbed browsing among mozilla firefox users","year":2010,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"World Wide Web; Computer science; Web browser; Web page; Web navigation; Information retrieval; The Internet","authors":[{"name":"Patrick Dubroy","is_ca":true},{"name":"Ravin Balakrishnan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0144848797964874,"gpt":0.2481044691568604,"spread":0.233619589360373,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002591301,0.0002893467,0.0004716159,0.001129852,0.001991138,0.001765935,0.0005136959,0.0009696203,0.00166099],"category_scores_gemma":[0.01041411,0.000531379,0.0002554103,0.0009037447,0.001104767,0.002276101,0.0007088421,0.000955345,0.0002895562],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005947134,"about_ca_system_score_gemma":0.0006048056,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005786799,"about_ca_topic_score_gemma":0.008536211,"domain_scores_codex":[0.9987183,0.000734305,0.00009093362,0.0001201428,0.0001634146,0.000172903],"domain_scores_gemma":[0.9880982,0.008426962,0.001179756,0.0004414705,0.0009481866,0.0009053511],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.0003874811,0.000737874,0.1863472,0.000290619,0.00003235058,0.001386777,0.7731162,0.00006474564,0.01277261,0.0003133437,0.0004046716,0.0241461],"study_design_scores_gemma":[0.00007646575,0.002598357,0.2621418,0.000201034,0.00005284539,0.002265354,0.7211533,0.001420189,0.002953198,0.0003255322,0.006706845,0.0001051215],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9994251,0.00004148769,0.0001750115,0.00004802643,0.00000158324,0.00001702174,0.00001234187,0.000005149141,0.0002743554],"genre_scores_gemma":[0.9988006,0.0001019216,0.0005424703,0.00008662283,0.000005165414,0.00003042836,0.00002050943,0.000005280178,0.0004070321],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005786799,"threshold_uncertainty_score":0.01370424,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2147293034","doi":"10.1145/2348283.2348327","title":"Mining query subtopics from search log data","year":2012,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Web search query; Information retrieval; Ranking (information retrieval); Cluster analysis; Web query classification; Search engine; Leverage (statistics); Data mining; Artificial intelligence","authors":[{"name":"Yunhua Hu","is_ca":false},{"name":"Yanan Qian","is_ca":false},{"name":"Hang Li","is_ca":false},{"name":"Daxin Jiang","is_ca":false},{"name":"Jian Pei","is_ca":true},{"name":"Qinghua Zheng","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1367392867182668,"gpt":0.3206255663048285,"spread":0.1838862795865617,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001574862,0.001743722,0.001608239,0.01242097,0.0009117637,0.001346564,0.001109283,0.001026755,0.0006121373],"category_scores_gemma":[0.008716805,0.0004018724,0.001827065,0.007518508,0.0005638376,0.002419779,0.001337046,0.00101287,0.0008979856],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009438854,"about_ca_system_score_gemma":0.001684619,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01619411,"about_ca_topic_score_gemma":0.02374642,"domain_scores_codex":[0.997203,0.0004446112,0.0003604622,0.0006172497,0.001064936,0.0003098988],"domain_scores_gemma":[0.9934781,0.00316849,0.0007368273,0.000688947,0.001614439,0.0003130892],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002480882,0.001199951,0.1925577,0.002852805,0.0005884253,0.002397171,0.003016622,0.03579283,0.1396458,0.004315716,0.02773596,0.5874162],"study_design_scores_gemma":[0.0001354591,0.0008225378,0.1919647,0.0002373808,0.0004510568,0.002746,0.003207405,0.7142742,0.05000475,0.01211362,0.02378086,0.0002619752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8021984,0.008368642,0.1519096,0.0007052484,0.0001209532,0.001007747,0.02495792,0.006583222,0.004148254],"genre_scores_gemma":[0.8446642,0.001475233,0.1086218,0.0001819829,0.0001412735,0.0003630951,0.04244083,0.0002359909,0.001875571],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01619411,"threshold_uncertainty_score":0.03219968,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2107131706","doi":"10.1109/icdm.2006.64","title":"Enhancing Text Clustering Using Concept-based Mining Model","year":2006,"lang":"en","type":"article","venue":"Proceedings","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Cluster analysis; Phrase; Sentence; Document clustering; Term (time); Semantics (computer science); Similarity (geometry); Natural language processing; Matching (statistics); Artificial intelligence; Similarity measure; Measure (data warehouse); Word (group theory); Information retrieval; Data mining; Mathematics; Image (mathematics)","authors":[{"name":"Shady Shehata","is_ca":true},{"name":"Fakhri Karray","is_ca":true},{"name":"Mohamed S. Kamel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02220185227204753,"gpt":0.2452552487338726,"spread":0.223053396461825,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002543673,0.0009222404,0.00152239,0.003132153,0.000723102,0.001466218,0.001887662,0.001246,0.001124006],"category_scores_gemma":[0.009291006,0.0003717512,0.001165754,0.003664048,0.0005518639,0.00331083,0.00109646,0.000997918,0.0008760668],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008505969,"about_ca_system_score_gemma":0.001251326,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00206702,"about_ca_topic_score_gemma":0.002152088,"domain_scores_codex":[0.9981452,0.0004657971,0.0001217934,0.0003266412,0.0008691802,0.00007133373],"domain_scores_gemma":[0.9959488,0.002042172,0.0002481069,0.0003281452,0.001367261,0.0000654908],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003824224,0.0008514712,0.004025596,0.0007997977,0.0003822511,0.0003963237,0.0008550308,0.1951983,0.03362295,0.02601971,0.007341005,0.7301251],"study_design_scores_gemma":[0.00003807523,0.0000990854,0.0005861864,0.00002483846,0.00007191855,0.0002761485,0.0001091435,0.971871,0.01034535,0.01296462,0.00357489,0.00003877798],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01794225,0.0002548921,0.9792842,0.0001721745,0.00004398574,0.0001875405,0.00009928611,0.0008893226,0.001126389],"genre_scores_gemma":[0.1177167,0.0003174047,0.8800128,0.0001165373,0.00006641315,0.0003179471,0.0004112952,0.00007115544,0.0009696566],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003132153,"threshold_uncertainty_score":0.01345241,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2056767790","doi":"10.1145/2508037.2508038","title":"Mining search and browse logs for web search","year":2013,"lang":"en","type":"article","venue":"ACM Transactions on Intelligent Systems and Technology","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Web search query; Terabyte; Web search engine; Search engine; Search analytics; Information retrieval; World Wide Web; Web crawler; Ranking (information retrieval); Web log analysis software; Transaction log; Metasearch engine; Database; Web page; Static web page; Web navigation","authors":[{"name":"Daxin Jiang","is_ca":false},{"name":"Jian Pei","is_ca":true},{"name":"Hang Li","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03611578898037965,"gpt":0.2776240375976158,"spread":0.2415082486172361,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002331513,0.001031143,0.001257645,0.01155495,0.0006983242,0.002165049,0.001283529,0.000967674,0.0007908831],"category_scores_gemma":[0.02317209,0.0005779457,0.001215251,0.008466845,0.0004119478,0.004377893,0.001118563,0.00132113,0.0012067],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006733637,"about_ca_system_score_gemma":0.001744636,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005932498,"about_ca_topic_score_gemma":0.01297555,"domain_scores_codex":[0.9978048,0.0005919825,0.0003134707,0.0003377998,0.0007961814,0.0001556797],"domain_scores_gemma":[0.9870617,0.007776345,0.001302263,0.001769589,0.001796325,0.0002938929],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007334998,0.00145883,0.2003548,0.002215625,0.0005400669,0.000888202,0.001778758,0.03315032,0.01354242,0.009022283,0.03115232,0.7051629],"study_design_scores_gemma":[0.00007272461,0.000323834,0.08191785,0.0002807823,0.0002811139,0.001276095,0.001662838,0.8479558,0.01288903,0.03313047,0.0200851,0.0001243955],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.486865,0.01047066,0.4411843,0.002202977,0.0001615174,0.001242817,0.03617948,0.01580373,0.00588961],"genre_scores_gemma":[0.6979474,0.003472952,0.2483506,0.0002334507,0.0002232196,0.0007238422,0.04615836,0.0004018934,0.002488328],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01155495,"threshold_uncertainty_score":0.01233035,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2287148974","doi":"10.1145/2888422.2888439","title":"Report on the SIGIR 2015 Workshop on Reproducibility, Inexplicability, and Generalizability of Results (RIGOR)","year":2016,"lang":"en","type":"article","venue":"ACM SIGIR Forum","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":59,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Università degli Studi di Padova","keywords":"Generalizability theory; Computer science; Presentation (obstetrics); Thursday; Information retrieval; Library science; Data science; Statistics; Medicine","authors":[{"name":"Jaime Arguello","is_ca":false},{"name":"Matt Crane","is_ca":false},{"name":"Fernando Díaz","is_ca":false},{"name":"Jimmy Lin","is_ca":true},{"name":"Andrew Trotman","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04259519423738254,"gpt":0.3006233049999692,"spread":0.2580281107625866,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4204637,0.004047017,0.00427287,0.007307969,0.007461053,0.02533596,0.008611387,0.009897928,0.03964162],"category_scores_gemma":[0.5271295,0.002115322,0.005136801,0.003938264,0.007451684,0.02916217,0.02470973,0.01356752,0.02502416],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007094688,"about_ca_system_score_gemma":0.01513458,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009034202,"about_ca_topic_score_gemma":0.006196078,"domain_scores_codex":[0.6590006,0.1876811,0.02007467,0.02524964,0.09905916,0.008935003],"domain_scores_gemma":[0.3206436,0.3235835,0.01790087,0.107462,0.2069122,0.02349778],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001918687,0.0007566003,0.00595755,0.002047982,0.0006152798,0.000409784,0.004064595,0.001457883,0.004444692,0.01352579,0.748414,0.2163873],"study_design_scores_gemma":[0.001091375,0.001509704,0.01334012,0.004033119,0.0008992302,0.0007777915,0.005072844,0.01136812,0.01424693,0.08823447,0.8586029,0.0008233882],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.02831216,0.06595184,0.3325501,0.3050986,0.1459833,0.008227564,0.01420851,0.01117791,0.08849011],"genre_scores_gemma":[0.2615598,0.02334031,0.3211905,0.08744568,0.06198272,0.01338972,0.06458687,0.01743254,0.1490718],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.5795363,"threshold_uncertainty_score":0.7146714,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W16404305","doi":"10.1097/01.prs.0000196262.13766.5c","title":"Proceedings of the 20th ACM conference on Hypertext and hypermedia","year":2009,"lang":"en","type":"article","venue":"ACM Conference on Hypertext","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":58,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Hypertext; Computer science; World Wide Web; Hypermedia; Presentation (obstetrics); Social media; Markup language; HTML; Multimedia; Web page; XML","authors":[{"name":"Ciro Cattuto","is_ca":false},{"name":"Giancarlo Ruffo","is_ca":false},{"name":"Filippo Menczer","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05153949935545338,"gpt":0.2611461444505319,"spread":0.2096066450950785,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003980011,0.00150376,0.002294923,0.00238601,0.001856375,0.01207637,0.002578441,0.003411057,0.2053416],"category_scores_gemma":[0.01172751,0.0006714484,0.001260825,0.002939827,0.002582848,0.01153531,0.005178593,0.005675543,0.1117799],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001728061,"about_ca_system_score_gemma":0.00379744,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003616783,"about_ca_topic_score_gemma":0.003948836,"domain_scores_codex":[0.9952602,0.001625933,0.0004047137,0.0008184279,0.001552621,0.0003381914],"domain_scores_gemma":[0.9922862,0.00256969,0.0003227051,0.001317339,0.002037663,0.00146636],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001413046,0.0001208647,0.0006602886,0.0005575408,0.00007705232,0.0002457924,0.0005401867,0.0002281629,0.002056435,0.01418062,0.816978,0.1642137],"study_design_scores_gemma":[0.00001134371,0.0000229131,0.0003177692,0.0001861686,0.00001710525,0.0001312289,0.0001111797,0.000371823,0.0002150142,0.003452532,0.9951479,0.00001505726],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.007962445,0.1252972,0.1264609,0.05096852,0.1381084,0.001564438,0.008396474,0.00819168,0.5330499],"genre_scores_gemma":[0.02118327,0.06099273,0.03361928,0.01060357,0.01757532,0.0008120117,0.0184493,0.002549613,0.8342149],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.7946584,"threshold_uncertainty_score":0.6869361,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2897050313","doi":"10.1145/3269206.3271728","title":"Personalizing Search Results Using Hierarchical RNN with Query-aware Attention","year":2018,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Personalization; Exploit; Information retrieval; Ranking (information retrieval); Search engine; Query expansion; Web search query; Personalized search; User information; Data mining; Quality (philosophy); World Wide Web","authors":[{"name":"Songwei Ge","is_ca":false},{"name":"Zhicheng Dou","is_ca":false},{"name":"Zhengbao Jiang","is_ca":false},{"name":"Jian‐Yun Nie","is_ca":true},{"name":"Ji-Rong Wen","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04730866175086899,"gpt":0.2962645276011868,"spread":0.2489558658503178,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001038655,0.0009080851,0.0007252904,0.000946373,0.0002100719,0.0004333197,0.00097635,0.0007292361,0.0008111243],"category_scores_gemma":[0.003265101,0.0004106543,0.0007001263,0.0008609626,0.0002958392,0.001088702,0.0004901754,0.0008635815,0.000442364],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001016545,"about_ca_system_score_gemma":0.0005760325,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01651397,"about_ca_topic_score_gemma":0.0186686,"domain_scores_codex":[0.9995492,0.00009244767,0.000029811,0.0001722956,0.00008028271,0.00007595355],"domain_scores_gemma":[0.9991295,0.0004692725,0.00009387571,0.00008727676,0.000177007,0.00004309415],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005822016,0.0005197946,0.01044056,0.000207805,0.0002316466,0.0001998466,0.0002794213,0.4486637,0.03321652,0.002070708,0.00423854,0.4993493],"study_design_scores_gemma":[0.000006453296,0.00003903668,0.000891621,0.000003341919,0.00001951092,0.0000197654,0.000006156821,0.9969023,0.001401057,0.0005765474,0.0001287634,0.000005392826],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3442621,0.001961115,0.6433402,0.0005499792,0.0001423731,0.0001921351,0.0004451069,0.004781292,0.004325696],"genre_scores_gemma":[0.9528741,0.0002885176,0.04407335,0.0001564502,0.00007055569,0.00006453434,0.0004032527,0.0000763442,0.001993018],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01651397,"threshold_uncertainty_score":0.03283572,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1924960892","doi":"10.3127/ajis.v7i2.270","title":"Issues of Page Representation and Organisation in Web Browser's Revisitation Tools","year":2000,"lang":"en","type":"article","venue":"AJIS. Australasian journal of information systems/AJIS. Australian journal of information systems/Australian journal of information systems","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada; Microsoft Research","keywords":"Representation (politics); World Wide Web; Computer science; Political science","authors":[{"name":"Andy Cockburn","is_ca":false},{"name":"Saul Greenberg","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02498188981610518,"gpt":0.2760008897835208,"spread":0.2510189999674157,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03224846,0.0009222685,0.00169306,0.003446711,0.002142964,0.01760741,0.00347772,0.003107283,0.003718888],"category_scores_gemma":[0.1843183,0.002000941,0.0009194898,0.004107096,0.00419096,0.0176387,0.004400397,0.003270101,0.002244944],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00174894,"about_ca_system_score_gemma":0.001825282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002859547,"about_ca_topic_score_gemma":0.002408977,"domain_scores_codex":[0.9745814,0.01587493,0.002949748,0.001449226,0.00431197,0.000832633],"domain_scores_gemma":[0.8114177,0.1303246,0.007461454,0.03173397,0.01700105,0.00206124],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001566552,0.0004801834,0.01228501,0.001916429,0.0001699839,0.000945356,0.04727258,0.01819694,0.02386927,0.2500691,0.03183291,0.6113958],"study_design_scores_gemma":[0.0005324363,0.001506352,0.02123306,0.004109088,0.0003965127,0.004797771,0.02139087,0.1226726,0.0475733,0.3030027,0.4713511,0.001434161],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07296149,0.003133283,0.8789099,0.00684031,0.0004378142,0.0004607785,0.0004661166,0.019791,0.0169993],"genre_scores_gemma":[0.4144358,0.002083126,0.562997,0.00110637,0.0002947286,0.0007215994,0.0005144567,0.007870028,0.009976917],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03224846,"threshold_uncertainty_score":0.1705481,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2148387158","doi":"10.1145/1557914.1557947","title":"Individual and social behavior in tagging systems","year":2009,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Pairwise comparison; Reuse; Similarity (geometry); Metric (unit); Information retrieval; Tag system; Recommender system; World Wide Web; Artificial intelligence","authors":[{"name":"Elizeu Santos‐Neto","is_ca":true},{"name":"David Condon","is_ca":false},{"name":"Nazareno Andrade","is_ca":false},{"name":"Adriana Iamnitchi","is_ca":false},{"name":"Matei Ripeanu","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02698872560099214,"gpt":0.2641396343212282,"spread":0.2371509087202361,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008612054,0.0004366433,0.0006651864,0.003070698,0.002103159,0.003680801,0.0008796154,0.001471869,0.001619548],"category_scores_gemma":[0.04264173,0.0005718624,0.0005277663,0.002850989,0.003186707,0.007255884,0.002377435,0.0009323388,0.0006235428],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001776025,"about_ca_system_score_gemma":0.0008097347,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003405388,"about_ca_topic_score_gemma":0.003314257,"domain_scores_codex":[0.9895329,0.005814731,0.0006943649,0.001958575,0.001534223,0.0004652792],"domain_scores_gemma":[0.9374906,0.04126967,0.007004481,0.007666373,0.004226576,0.002342216],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0008063021,0.0007834884,0.6918293,0.0004487551,0.0005825469,0.0008477778,0.02493045,0.0386548,0.02375675,0.06530495,0.001758074,0.1502969],"study_design_scores_gemma":[0.00009313605,0.0007379232,0.3781594,0.00008927495,0.0002731065,0.001819415,0.01140899,0.4001929,0.01345133,0.1739598,0.01942211,0.0003925739],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8934022,0.000239761,0.09844465,0.0004782347,0.00001635715,0.0001689483,0.0002448896,0.0003749533,0.006629915],"genre_scores_gemma":[0.9855146,0.00005944611,0.01300456,0.00004878906,0.00001668998,0.00008368037,0.0001813365,0.0000625297,0.001028372],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008612054,"threshold_uncertainty_score":0.0455454,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4285540454","doi":"10.52458/978-93-91842-08-6-38","title":"Web Scraping Techniques and Applications: A Literature Review","year":2021,"lang":"en","type":"review","venue":"Soft Computing Research Society eBooks","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Cégep de Chicoutimi; Université du Québec à Chicoutimi","funders":"","keywords":"Computer science; World Wide Web; Web intelligence; Big data; Data science; Web mining; Web analytics; Web application; Analytics; Social media; Web standards; Web development; Web engineering; Web modeling; The Internet; Web page; Data mining","authors":[{"name":"Chaimaa Lotfi","is_ca":true},{"name":"Swetha Srinivasan","is_ca":true},{"name":"Myriam Ertz","is_ca":true},{"name":"Imen Latrous","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0854457235976239,"gpt":0.4236613379830667,"spread":0.3382156143854428,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002049687,0.00107107,0.001560047,0.01008933,0.0007076002,0.002079927,0.001390041,0.001316542,0.00397356],"category_scores_gemma":[0.006809024,0.0006805962,0.001316573,0.01237519,0.0007800526,0.003373937,0.000975812,0.001318444,0.001701263],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009278256,"about_ca_system_score_gemma":0.003572143,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003310082,"about_ca_topic_score_gemma":0.004924365,"domain_scores_codex":[0.9987337,0.0002478078,0.000248946,0.00016816,0.0005303852,0.00007093301],"domain_scores_gemma":[0.9933404,0.004302704,0.0005651999,0.000137198,0.001515771,0.0001386954],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004473941,0.00009931128,0.0005179359,0.05352835,0.0001120058,0.0001905323,0.0002150855,0.0003964326,0.0006128149,0.00251033,0.01480396,0.9269685],"study_design_scores_gemma":[0.00001855064,0.0002047485,0.003800734,0.07926469,0.0007579902,0.002645392,0.0006714191,0.0006114329,0.001398381,0.004310669,0.9062393,0.00007657528],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0003033988,0.9976619,0.0004918237,0.0003452962,0.0001181985,0.00001807219,0.00002722492,0.00001658808,0.001017369],"genre_scores_gemma":[0.00135018,0.9972824,0.0007395396,0.0001826957,0.0001020533,0.00001915751,0.00003818344,0.000005478057,0.0002803302],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.01008933,"threshold_uncertainty_score":0.01329291,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2169558036","doi":"10.1109/dnsr.2004.1344744","title":"The reconstruction of user sessions from a server log using improved time-oriented heuristics","year":2004,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of New Brunswick","funders":"Atlantic Canada Opportunities Agency","keywords":"Computer science; Heuristics; Web log analysis software; Web mining; Web server; Personalization; Web page; Web modeling; Web service; Clickstream; Data Web; World Wide Web; Web navigation; Adaptation (eye); Information retrieval; Data mining; Static web page; Web API; The Internet","authors":[{"name":"J. Zhang","is_ca":true},{"name":"Ali A. Ghorbani","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01334641712887633,"gpt":0.2362877217104381,"spread":0.2229413045815618,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002413941,0.001005772,0.001458549,0.002388077,0.0005705187,0.001258451,0.001747754,0.0009250829,0.0008286531],"category_scores_gemma":[0.01462446,0.0005644748,0.001004841,0.002166286,0.0006469832,0.001921447,0.0007886337,0.0008825605,0.0003280365],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001182087,"about_ca_system_score_gemma":0.003124728,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01033898,"about_ca_topic_score_gemma":0.01085926,"domain_scores_codex":[0.9973699,0.001000709,0.0002387915,0.0004771026,0.0006182209,0.0002953123],"domain_scores_gemma":[0.9875885,0.008014772,0.001118349,0.001429803,0.00136481,0.0004838659],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001050947,0.0007631638,0.01902782,0.0002448884,0.0002582894,0.0002426422,0.0006921047,0.6283032,0.01325189,0.008034924,0.002303786,0.3258262],"study_design_scores_gemma":[0.00003815507,0.00008717447,0.001538994,0.000006840677,0.0000351634,0.00007549763,0.00006699444,0.9928976,0.002476307,0.002293641,0.000460625,0.00002296271],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1192015,0.0002790579,0.8774001,0.0001059233,0.00002320683,0.0001840363,0.0002275483,0.001855776,0.0007228588],"genre_scores_gemma":[0.404364,0.0001531825,0.5933832,0.00006627006,0.00003066271,0.0002323596,0.0009889359,0.0002144662,0.0005669669],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01033898,"threshold_uncertainty_score":0.02055764,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2121415061","doi":"10.1109/iv.2006.108","title":"The Visual Exploration ofWeb Search Results Using HotMap","year":2006,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Regina","funders":"","keywords":"Computer science; Information retrieval; Search engine; Relevance (law); Set (abstract data type); Representation (politics); Result set; World Wide Web; Web search query; Usability; Search analytics; Visual search; Web search engine; Human–computer interaction; Artificial intelligence","authors":[{"name":"Orland Hoeber","is_ca":true},{"name":"Xue Dong Yang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04998496559217664,"gpt":0.3085191218787094,"spread":0.2585341562865328,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001117316,0.001225469,0.0005595824,0.003877623,0.0004043365,0.00239282,0.0008043326,0.0008047401,0.01510796],"category_scores_gemma":[0.00425005,0.0004013781,0.0007029424,0.002365823,0.0004075622,0.003012836,0.003133383,0.0007250583,0.00302621],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002880863,"about_ca_system_score_gemma":0.0003929126,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001234251,"about_ca_topic_score_gemma":0.001899344,"domain_scores_codex":[0.9995185,0.0001682744,0.00002620817,0.00004989122,0.0001815381,0.00005554992],"domain_scores_gemma":[0.9980146,0.001175275,0.000106441,0.0001977514,0.0003334376,0.0001724799],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003616526,0.0003859712,0.005103339,0.004412219,0.0003111646,0.00244598,0.009292649,0.01750297,0.06602743,0.02966002,0.1808645,0.6803772],"study_design_scores_gemma":[0.0007243792,0.001268091,0.01994942,0.0016841,0.0003416438,0.005473693,0.005782199,0.3409488,0.0664032,0.1036709,0.45312,0.0006334881],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09554125,0.005283177,0.7761848,0.002112751,0.0005759306,0.0007536608,0.009183022,0.05235628,0.05800919],"genre_scores_gemma":[0.3884403,0.003534739,0.5787446,0.0008598398,0.0003351993,0.0007794226,0.006831602,0.003982379,0.01649189],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01510796,"threshold_uncertainty_score":0.05054116,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1554796723","doi":"10.1007/3-540-46105-1_24","title":"Finding Similar Queries to Satisfy Searches Based on Query Traces","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Information retrieval; Web query classification; Search engine; Web search query; Similarity (geometry); Query expansion; Task (project management); Spatial query; Disappointment; Query language; Sargable; Query optimization; Variety (cybernetics); Search-oriented architecture; Ranking (information retrieval); Data mining; Artificial intelligence","authors":[{"name":"Osmar R. Zai͏̈ane","is_ca":true},{"name":"Alexander Strilets","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03937416903002822,"gpt":0.2637484585050917,"spread":0.2243742894750635,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00335404,0.001175266,0.002311336,0.006860778,0.001528568,0.003826431,0.00227443,0.002282555,0.002347135],"category_scores_gemma":[0.03661977,0.0008740599,0.001676645,0.005087417,0.001189815,0.007966313,0.002271103,0.002100555,0.0009987768],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001117239,"about_ca_system_score_gemma":0.003016572,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005495312,"about_ca_topic_score_gemma":0.008153332,"domain_scores_codex":[0.9932921,0.0009499323,0.0008641247,0.0009868486,0.003230663,0.0006763145],"domain_scores_gemma":[0.9700415,0.01894197,0.001589775,0.003827801,0.004220575,0.001378481],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.006763405,0.002959967,0.198836,0.00238862,0.001528475,0.002629296,0.004774726,0.0680868,0.1586147,0.0358534,0.0215,0.4960647],"study_design_scores_gemma":[0.000218338,0.000930008,0.01617021,0.0001186188,0.0006466759,0.002161176,0.002330322,0.8823588,0.04429063,0.04235763,0.008287021,0.0001306135],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.7461585,0.00162645,0.2362336,0.001809482,0.0001408061,0.0007322758,0.003294163,0.006617558,0.003387102],"genre_scores_gemma":[0.8563855,0.0005289022,0.1330469,0.0002888143,0.0001424419,0.0001877736,0.006816828,0.0006814296,0.001921405],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.006860778,"threshold_uncertainty_score":0.0177381,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2806438009","doi":"10.1007/978-3-540-85902-4_35","title":"University of Waterloo at INEX2007: Adhoc and Link-the-Wiki Tracks","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Link (geometry); XML; Embedding; Track (disk drive); Context (archaeology); Point (geometry); Information retrieval; World Wide Web; Computer network; Artificial intelligence; Operating system","authors":[{"name":"Kelly Y. Itakura","is_ca":true},{"name":"Charles L. A. Clarke","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01357757496992636,"gpt":0.1988799585418461,"spread":0.1853023835719198,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002930558,0.001039042,0.001138129,0.004055322,0.002798127,0.00546907,0.001688124,0.001017386,0.1363685],"category_scores_gemma":[0.005831896,0.0008341069,0.0004412368,0.008213246,0.0005836933,0.006272604,0.002049976,0.001467102,0.06827516],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00331049,"about_ca_system_score_gemma":0.009096751,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.1177996,"about_ca_topic_score_gemma":0.2055963,"domain_scores_codex":[0.9980729,0.0001732429,0.00006512737,0.0002773926,0.001197843,0.000213497],"domain_scores_gemma":[0.99546,0.0004441982,0.0001093351,0.0005291447,0.002188259,0.001269169],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004081814,0.00007284785,0.0003734943,0.00008227738,0.000004411404,0.00002265104,0.00009735884,0.0002431483,0.0004199632,0.003685337,0.8612062,0.1337515],"study_design_scores_gemma":[0.00001500801,0.00002252001,0.001271766,0.00005115059,0.000004936236,0.00004434713,0.0000920664,0.001499478,0.001326892,0.002099692,0.9935529,0.00001925388],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.01897078,0.01800124,0.07299695,0.02327336,0.01121596,0.0005511136,0.1610695,0.0595964,0.6343247],"genre_scores_gemma":[0.01784087,0.004667351,0.04242847,0.0008839474,0.001182723,0.0001540175,0.09465922,0.009482441,0.8287009],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.1363685,"threshold_uncertainty_score":0.4561981,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1570003767","doi":"","title":"Predicting Future User Actions by Observing Unmodified Applications","year":2000,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; User interface; User modeling; Human–computer interaction; State space; Software; User interface design; State (computer science); Space (punctuation); Machine learning; Artificial intelligence; Distributed computing; Data mining; User experience design; Algorithm; Programming language","authors":[{"name":"Peter Gorniak","is_ca":true},{"name":"David Poole","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01742366038573868,"gpt":0.2458363644001544,"spread":0.2284127040144157,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008818423,0.0008232936,0.0005575173,0.000872933,0.0002840328,0.000989634,0.0006939668,0.0008376513,0.0008680372],"category_scores_gemma":[0.007970376,0.00053328,0.0004238629,0.0005931394,0.0004117988,0.001801612,0.000624157,0.001042173,0.0006136617],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004799906,"about_ca_system_score_gemma":0.00063305,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008221854,"about_ca_topic_score_gemma":0.01597537,"domain_scores_codex":[0.99936,0.0001980156,0.00004484861,0.0001647367,0.0001895,0.00004291148],"domain_scores_gemma":[0.995394,0.002747712,0.0005372946,0.000664453,0.0005311413,0.0001254107],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008580034,0.0008609656,0.1900067,0.0002345642,0.0002234233,0.0002845737,0.0008636729,0.3850237,0.02571281,0.002388838,0.003931128,0.3896116],"study_design_scores_gemma":[0.000008402209,0.0001003567,0.01396485,0.000007212868,0.00002731829,0.00004744571,0.00005397907,0.9792879,0.004033335,0.001860046,0.0005893988,0.00001985764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6601502,0.0002940223,0.3312081,0.0003403667,0.00002717977,0.0001753069,0.0006936666,0.004679373,0.002431852],"genre_scores_gemma":[0.9292902,0.0001852819,0.06883154,0.00004659596,0.00001466527,0.00006973535,0.0006866812,0.00007805075,0.000797282],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008221854,"threshold_uncertainty_score":0.016348,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W15738104","doi":"10.1145/2488388.2488459","title":"Imagen","year":2013,"lang":"es","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; JavaScript; HTML5; Session (web analytics); Web browser; Rich Internet application; World Wide Web; Client-side scripting; Snapshot (computer storage); Operating system; Web application; Multimedia; Web page; Web API; The Internet; Web navigation","authors":[{"name":"James Teng Kin Lo","is_ca":true},{"name":"Eric Wohlstadter","is_ca":true},{"name":"Ali Mesbah","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01083153206537008,"gpt":0.2287507717079443,"spread":0.2179192396425742,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0003954006,0.001000439,0.0006569204,0.002070615,0.0008337179,0.004332697,0.001474696,0.002103751,0.8220645],"category_scores_gemma":[0.001787478,0.0003927613,0.0006097904,0.001517338,0.0005296097,0.002205955,0.001920764,0.001257721,0.6563246],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009707698,"about_ca_system_score_gemma":0.0007697248,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002944968,"about_ca_topic_score_gemma":0.00325229,"domain_scores_codex":[0.9995241,0.00004439542,0.00002833367,0.000114893,0.0002396115,0.0000486732],"domain_scores_gemma":[0.9992544,0.0001096002,0.00004221427,0.0001686966,0.0002813213,0.0001437004],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002213244,0.00006713797,0.0005681433,0.0003681786,0.00001515141,0.0004363361,0.00008675002,0.0002625541,0.003909767,0.007495695,0.6301968,0.3563722],"study_design_scores_gemma":[0.00001837567,0.00001443133,0.0004525975,0.00007577635,0.000004734874,0.0004441543,0.00003915857,0.0004611309,0.0007825224,0.001100386,0.9965993,0.000007404622],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.001612634,0.002082309,0.009456213,0.002257104,0.002730081,0.0002146229,0.007915871,0.00771252,0.9660186],"genre_scores_gemma":[0.01097974,0.001926713,0.006630129,0.001560991,0.0006893358,0.0001177007,0.009151052,0.002022479,0.9669219],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.1779355,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2793625150","doi":"10.1002/asi.24048","title":"If these crawls could talk: Studying and documenting web archives provenance","year":2018,"lang":"en","type":"article","venue":"Journal of the Association for Information Science and Technology","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo; University of Toronto","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; University of Toronto","keywords":"World Wide Web; Computer science; Transparency (behavior); Context (archaeology); Data science; Process (computing); Geography; Archaeology","authors":[{"name":"Emily Maemura","is_ca":true},{"name":"Nicholas Worby","is_ca":true},{"name":"Ian Milligan","is_ca":true},{"name":"Christoph Becker","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.009031874620079431,"gpt":0.2572771037646895,"spread":0.2482452291446101,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.02888314,0.0003587739,0.0005278946,0.00808701,0.006125059,0.01065745,0.001260891,0.001457262,0.001757334],"category_scores_gemma":[0.148514,0.000952763,0.0006184802,0.006582465,0.006081573,0.01454659,0.005231049,0.002092135,0.0005452758],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003513149,"about_ca_system_score_gemma":0.006998329,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01201761,"about_ca_topic_score_gemma":0.01849698,"domain_scores_codex":[0.9790562,0.01231174,0.001875873,0.001573573,0.004526153,0.0006564723],"domain_scores_gemma":[0.8665952,0.08960489,0.01252298,0.01847989,0.01144295,0.001354112],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0005412119,0.0002908982,0.2166717,0.0008106907,0.0001286975,0.002244584,0.4190868,0.002723923,0.006205402,0.1183854,0.00431576,0.2285949],"study_design_scores_gemma":[0.0001070045,0.0003358926,0.1818789,0.002890706,0.0004034496,0.003085685,0.2674572,0.02474111,0.02687773,0.2069003,0.2848721,0.0004500595],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8341457,0.002111535,0.1106026,0.003947759,0.0002119672,0.001108731,0.001301015,0.001107324,0.04546328],"genre_scores_gemma":[0.9164476,0.001092215,0.07524218,0.0002649099,0.00007829519,0.0004572193,0.0007548556,0.0004432504,0.005219341],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9893426,"threshold_uncertainty_score":0.1527505,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2034607275","doi":"10.1145/1046456.1046459","title":"Learning important models for web page blocks based on layout and content analysis","year":2004,"lang":"en","type":"article","venue":"ACM SIGKDD Explorations Newsletter","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Web page; Block (permutation group theory); Information retrieval; Partition (number theory); Page view; Static web page; Data mining; World Wide Web; Web navigation","authors":[{"name":"Ruihua Song","is_ca":false},{"name":"Haifeng Liu","is_ca":true},{"name":"Ji-Rong Wen","is_ca":false},{"name":"Wei‐Ying Ma","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07156513468895788,"gpt":0.263723749372729,"spread":0.1921586146837712,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00182862,0.001169123,0.001306003,0.004589635,0.0004590129,0.001424245,0.001739314,0.001763806,0.001474455],"category_scores_gemma":[0.00860346,0.0008015702,0.001885037,0.001680219,0.0008838507,0.003108758,0.0007338889,0.002000357,0.0009379449],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001595592,"about_ca_system_score_gemma":0.0007554874,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006761526,"about_ca_topic_score_gemma":0.01195777,"domain_scores_codex":[0.9991167,0.0002434251,0.00006878802,0.0002612631,0.0002004821,0.000109279],"domain_scores_gemma":[0.9950523,0.003272102,0.0004518704,0.0003802472,0.0006836073,0.0001599412],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001016153,0.0008458124,0.02730521,0.0003148658,0.0002784122,0.0003444196,0.0004048986,0.621131,0.0125158,0.01028108,0.005888219,0.3196743],"study_design_scores_gemma":[0.00000810925,0.0000224061,0.0008096609,0.000006287091,0.00001491555,0.0000302837,0.000010084,0.9954496,0.0007566248,0.002700908,0.0001859514,0.000005177646],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1488709,0.0007031263,0.8453839,0.0003921349,0.00006289181,0.0002692192,0.0005604704,0.002684837,0.001072639],"genre_scores_gemma":[0.7508843,0.0004855798,0.2420981,0.000170843,0.0001511158,0.0004258105,0.00228132,0.0002894324,0.00321359],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006761526,"threshold_uncertainty_score":0.0134443,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2004193186","doi":"10.1145/584931.584940","title":"A framework for web table mining","year":2002,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Table (database); Web page; Information retrieval; World Wide Web; Web mining; Table of contents; Information extraction; Process (computing); Data mining","authors":[{"name":"Yingchen Yang","is_ca":true},{"name":"Wo-Shun Luk","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04721870685743441,"gpt":0.2677927006460024,"spread":0.220573993788568,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008295611,0.001865742,0.001893133,0.009678522,0.002424001,0.008419906,0.006132294,0.002592888,0.005897294],"category_scores_gemma":[0.02135476,0.001689631,0.005261851,0.01001889,0.003422442,0.009660353,0.004774561,0.00390068,0.004698471],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001853904,"about_ca_system_score_gemma":0.003942176,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007201462,"about_ca_topic_score_gemma":0.008264365,"domain_scores_codex":[0.9912164,0.002539047,0.001144468,0.001429641,0.003341884,0.0003285606],"domain_scores_gemma":[0.9902029,0.004705448,0.0007015966,0.002436866,0.001570432,0.0003826529],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00008990621,0.0002488307,0.002647809,0.001042554,0.0002775374,0.000910937,0.001074759,0.03867513,0.002343336,0.5923201,0.03732972,0.3230394],"study_design_scores_gemma":[0.00004708382,0.00005812757,0.0006027419,0.0003705869,0.00007639646,0.0009875651,0.0003315989,0.2066089,0.002091391,0.6479988,0.1407405,0.00008641461],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0003530056,0.0003522526,0.9953374,0.0003607813,0.00004295688,0.0002504944,0.0006239568,0.001637692,0.001041447],"genre_scores_gemma":[0.006604623,0.0003582253,0.9901085,0.0001288751,0.00006463225,0.0003205226,0.001514374,0.0001442571,0.0007560035],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009678522,"threshold_uncertainty_score":0.04387188,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W130931946","doi":"10.1007/978-1-4615-1507-4_16","title":"Recent Advances in Tabu Search","year":2002,"lang":"en","type":"book-chapter","venue":"Operations research, computer science. Interface series","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal; Université du Québec à Montréal","funders":"","keywords":"Tabu search; Guided Local Search; Focus (optics); Computer science; Field (mathematics); Hill climbing; Artificial intelligence; Mathematics; Physics","authors":[{"name":"Michel Gendreau","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07058871474709118,"gpt":0.3526118933868006,"spread":0.2820231786397094,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003721417,0.001289254,0.002906293,0.005080382,0.0009764889,0.004178212,0.003703363,0.001588272,0.02713589],"category_scores_gemma":[0.01610247,0.00086084,0.001183973,0.02122301,0.001584215,0.006448261,0.001844727,0.00232538,0.01213105],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002028876,"about_ca_system_score_gemma":0.003051594,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005022422,"about_ca_topic_score_gemma":0.004671318,"domain_scores_codex":[0.9963351,0.001339244,0.0002321099,0.0005480865,0.001326364,0.0002190473],"domain_scores_gemma":[0.992074,0.005050564,0.000367148,0.0009661713,0.001351057,0.0001910357],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000194298,0.0001055896,0.0005530968,0.002034505,0.00007374532,0.00002437584,0.00007083278,0.007569348,0.0003943316,0.04288146,0.04531183,0.9007866],"study_design_scores_gemma":[0.0001833785,0.0003134083,0.001724981,0.0016533,0.000291393,0.0009909137,0.0002345555,0.1306635,0.002777042,0.2187944,0.6422328,0.0001403345],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.005121503,0.6947418,0.2353345,0.003798986,0.003160042,0.0001541195,0.001136899,0.0036124,0.05293973],"genre_scores_gemma":[0.07139518,0.3716766,0.5030908,0.003308069,0.007447483,0.0004294142,0.005009834,0.001959317,0.03568339],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.02713589,"threshold_uncertainty_score":0.09077853,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2167213883","doi":"10.1145/2047196.2047224","title":"Query-feature graphs","year":2011,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; USable; Web query classification; Feature (linguistics); Information retrieval; Graphics; Web search query; Search engine; Theoretical computer science; World Wide Web; Computer graphics (images)","authors":[{"name":"Adam Fourney","is_ca":true},{"name":"Richard Mann","is_ca":true},{"name":"Michael Terry","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02612589293753511,"gpt":0.2131534380683716,"spread":0.1870275451308365,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032092,0.001283427,0.0008494185,0.003988916,0.001626845,0.004048395,0.002579927,0.001975605,0.01372746],"category_scores_gemma":[0.01936449,0.0008224835,0.001852984,0.004588518,0.002262076,0.01192883,0.003376677,0.00201448,0.003368684],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001824859,"about_ca_system_score_gemma":0.001651366,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009941251,"about_ca_topic_score_gemma":0.00904359,"domain_scores_codex":[0.99537,0.00121655,0.0004469105,0.001029083,0.00156404,0.0003734871],"domain_scores_gemma":[0.9896542,0.005194105,0.0005305707,0.00257912,0.001699028,0.0003430382],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002370097,0.0001535287,0.002824425,0.0006087932,0.00008824174,0.0007673587,0.001576043,0.01971566,0.004241255,0.7463613,0.04009663,0.1833297],"study_design_scores_gemma":[0.00003746162,0.0000893024,0.0006335863,0.0001500909,0.00006243928,0.0007044699,0.0005232106,0.07363593,0.00530256,0.6535125,0.2652515,0.00009693397],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005129221,0.0003966101,0.9714116,0.0009231092,0.0001128405,0.0004027533,0.004286741,0.006386472,0.01095068],"genre_scores_gemma":[0.1751537,0.001012295,0.7885288,0.001417213,0.000152707,0.00109383,0.01455624,0.003154937,0.01493036],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01372746,"threshold_uncertainty_score":0.04592294,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1870211331","doi":"10.1007/11944935_13","title":"Text Mining Through Semi Automatic Semantic Annotation","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto; Queen's University","funders":"","keywords":"Annotation; Computer science; Semantic annotation; Information retrieval; Natural language processing; Artificial intelligence","authors":[{"name":"Nadzeya Kiyavitskaya","is_ca":false},{"name":"Nicola Zeni","is_ca":false},{"name":"Luisa Mich","is_ca":false},{"name":"James R. Cordy","is_ca":true},{"name":"John Mylopoulos","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01645471353785901,"gpt":0.2481572614898082,"spread":0.2317025479519492,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001615149,0.001022826,0.0009638499,0.004957539,0.001106887,0.002332371,0.001268537,0.0007698136,0.00435577],"category_scores_gemma":[0.005742403,0.0006111308,0.00127395,0.00455482,0.0009666023,0.003906933,0.002224858,0.001251344,0.005918967],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004468421,"about_ca_system_score_gemma":0.00129645,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001318258,"about_ca_topic_score_gemma":0.002125006,"domain_scores_codex":[0.99817,0.0004685085,0.0002264198,0.0004006755,0.0006536517,0.00008068245],"domain_scores_gemma":[0.9954136,0.002326246,0.0002521776,0.0007580759,0.001179059,0.00007093616],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000341306,0.0001835249,0.00140098,0.0008061239,0.0001156245,0.0003475191,0.0006189868,0.004760188,0.05173935,0.02545107,0.02537156,0.8888637],"study_design_scores_gemma":[0.00008279605,0.0002367232,0.00318558,0.0004221566,0.0003935393,0.001520912,0.001013251,0.4768323,0.1797101,0.19418,0.1422713,0.0001512207],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01377144,0.0005577868,0.9656375,0.0003761479,0.0002090038,0.0002359069,0.002622962,0.01124878,0.005340517],"genre_scores_gemma":[0.1165153,0.0007786226,0.8576313,0.0002048194,0.0001767418,0.0005209094,0.01416109,0.001184394,0.008826699],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004957539,"threshold_uncertainty_score":0.01457149,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2110780226","doi":"10.1145/1031453.1031458","title":"Probabilistic models for focused web crawling","year":2004,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"Mitacs","keywords":"CRFS; Crawling; Computer science; Conditional random field; Hidden Markov model; Web crawler; Probabilistic logic; Context (archaeology); Relevance (law); Artificial intelligence; Web page; Machine learning; Information retrieval; Data mining; World Wide Web","authors":[{"name":"Hongyu Liu","is_ca":true},{"name":"Evangelos Milios","is_ca":true},{"name":"Jeannette Janssen","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03819472388232283,"gpt":0.2513862193380534,"spread":0.2131914954557306,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003416491,0.0009118593,0.001322994,0.002919164,0.0008394952,0.002074781,0.002630862,0.0023953,0.003583713],"category_scores_gemma":[0.01888295,0.00135622,0.001622349,0.002771269,0.0008107136,0.003769011,0.00110517,0.001801885,0.001767436],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001294737,"about_ca_system_score_gemma":0.001381719,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01110845,"about_ca_topic_score_gemma":0.01399539,"domain_scores_codex":[0.9979795,0.0007983723,0.0001587919,0.0004190495,0.0005387891,0.0001054887],"domain_scores_gemma":[0.9877757,0.009223141,0.0006189485,0.001276392,0.0009460825,0.0001596842],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009509427,0.0001132189,0.001915509,0.0001827391,0.0001345593,0.0002233686,0.0002498979,0.8500144,0.001240219,0.06529982,0.006349663,0.0741815],"study_design_scores_gemma":[0.00001160885,0.000008815004,0.0002431749,0.00001000081,0.00001461418,0.00004778799,0.000008449565,0.9641122,0.0001928712,0.03371776,0.001616699,0.00001606944],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005623993,0.0003830525,0.9906483,0.0001994298,0.00002721714,0.00006482,0.000469889,0.001583442,0.0009999096],"genre_scores_gemma":[0.2536176,0.001748168,0.7343921,0.0002613276,0.0002387991,0.0009720279,0.003065038,0.0006217817,0.0050832],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01110845,"threshold_uncertainty_score":0.02208757,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2963451689","doi":"","title":"SocialScope: Enabling Information Discovery on Social Content Sites","year":2009,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; World Wide Web; Presentation (obstetrics); Data science; Information sharing; Content management; Knowledge management","authors":[{"name":"Sihem Amer-Yahia","is_ca":false},{"name":"Laks V. S. Lakshmanan","is_ca":true},{"name":"Cong Yu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05891721385290173,"gpt":0.2718646654349444,"spread":0.2129474515820426,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005611917,0.0006363692,0.0008432532,0.003331698,0.002880103,0.006937052,0.003142182,0.001600929,0.004411747],"category_scores_gemma":[0.01530437,0.001010312,0.001636682,0.004604047,0.004339952,0.01405222,0.008920477,0.002346177,0.00172726],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002172336,"about_ca_system_score_gemma":0.003685896,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006690759,"about_ca_topic_score_gemma":0.00947425,"domain_scores_codex":[0.9940713,0.002218348,0.0004931487,0.0007593831,0.002064581,0.0003931717],"domain_scores_gemma":[0.9918597,0.0032556,0.0006332854,0.002651325,0.001080565,0.0005195243],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001405743,0.0001732971,0.002639058,0.0003057512,0.00007792689,0.0003391588,0.0007171177,0.02786232,0.006164516,0.8368026,0.01249759,0.11228],"study_design_scores_gemma":[0.0001003487,0.0001190362,0.0009756802,0.0001092868,0.00008595043,0.0004826772,0.0005422861,0.4323016,0.01459528,0.4410073,0.1095806,0.00009995158],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007284943,0.0001576149,0.9825156,0.0008247097,0.00004051369,0.0002962484,0.0002776681,0.002697592,0.005905091],"genre_scores_gemma":[0.1538054,0.0005217774,0.83866,0.0001939174,0.00009991496,0.0004059228,0.001147391,0.0002862298,0.004879454],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006937052,"threshold_uncertainty_score":0.029679,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W23637940","doi":"10.1371/journal.pone.0061981","title":"Crawling rich internet applications: the state of the art","year":2012,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Crawling; Computer science; Ajax; World Wide Web; Web crawler; The Internet; Web application; Asynchronous communication; Rich Internet application; State (computer science); Server; Multimedia; Telecommunications","authors":[{"name":"Suryakant Choudhary","is_ca":true},{"name":"Mustafa Emre Dinçtürk","is_ca":true},{"name":"Seyed M. Mirtaheri","is_ca":true},{"name":"Ali Moosavi","is_ca":true},{"name":"Gregor von Bochmann","is_ca":true},{"name":"Guy-Vincent Jourdan","is_ca":true},{"name":"Iosif Viorel Onut","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04276188563729921,"gpt":0.2296142188181238,"spread":0.1868523331808246,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005603783,0.001837844,0.003602206,0.01236728,0.001448735,0.01343548,0.004966131,0.004140247,0.002452918],"category_scores_gemma":[0.03296741,0.00210405,0.001618452,0.01569673,0.003520835,0.01724455,0.002948873,0.002001035,0.002873661],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001290965,"about_ca_system_score_gemma":0.002273544,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009247414,"about_ca_topic_score_gemma":0.006082972,"domain_scores_codex":[0.9907117,0.001980136,0.0007919742,0.002597424,0.003624406,0.0002943154],"domain_scores_gemma":[0.9375094,0.04158517,0.002657919,0.01016001,0.007092257,0.0009952594],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001465946,0.0003175476,0.01941858,0.005384276,0.0003961176,0.0001049936,0.00124024,0.008594858,0.002461281,0.01521704,0.01814413,0.9285743],"study_design_scores_gemma":[0.0001587049,0.0006418446,0.06173541,0.008127819,0.001013043,0.002365974,0.006394217,0.3873126,0.01570255,0.1089438,0.407017,0.0005869829],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.1555713,0.4674946,0.3037893,0.01541422,0.001132071,0.000952484,0.006699489,0.01231095,0.03663545],"genre_scores_gemma":[0.4240176,0.3047048,0.238122,0.002793186,0.003681002,0.0007191962,0.01553022,0.001771899,0.00865997],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.01343548,"threshold_uncertainty_score":0.02963603,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4251596819","doi":"10.1007/978-3-540-78646-7_22","title":"Automatic Extraction of Domain-Specific Stopwords from Labeled Documents","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Classifier (UML); Information loss; Filter (signal processing); Data mining; Set (abstract data type); Artificial intelligence; Information retrieval","authors":[{"name":"Masoud Makrehchi","is_ca":true},{"name":"Mohamed S. Kamel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01686015710638942,"gpt":0.2507353239094569,"spread":0.2338751668030674,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007644778,0.002187424,0.001986004,0.006268492,0.001396435,0.00250564,0.001379548,0.001840924,0.00230274],"category_scores_gemma":[0.003164721,0.0007532107,0.001246348,0.004833257,0.0004474839,0.00154096,0.0009874395,0.001640988,0.007610884],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006389024,"about_ca_system_score_gemma":0.00229727,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002476505,"about_ca_topic_score_gemma":0.005310946,"domain_scores_codex":[0.998919,0.0001417362,0.0001737856,0.0002539677,0.0003906132,0.0001210055],"domain_scores_gemma":[0.9953095,0.001508193,0.0004611314,0.0003518362,0.002210665,0.0001587206],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007764255,0.00041706,0.006500702,0.003011819,0.0002641183,0.002741187,0.0006736444,0.002585668,0.2917601,0.002570155,0.04176809,0.646931],"study_design_scores_gemma":[0.0002522324,0.000784207,0.0280547,0.0008588358,0.001352571,0.007786584,0.001984518,0.2016178,0.6168149,0.01039246,0.1298112,0.0002899544],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2388124,0.0090813,0.6758901,0.001081767,0.001437336,0.001370654,0.0222636,0.03889545,0.01116738],"genre_scores_gemma":[0.2378403,0.003924037,0.658913,0.0005011,0.0004083557,0.0006231207,0.08146454,0.002116113,0.01420954],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006268492,"threshold_uncertainty_score":0.007703424,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2166367071","doi":"10.1109/wi.2006.103","title":"Interactive Web Information Retrieval Using WordBars","year":2006,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Regina","funders":"","keywords":"Computer science; Information retrieval; Web query classification; Web search query; Query expansion; Set (abstract data type); sort; Result set; Search engine; Query language; World Wide Web; Information needs","authors":[{"name":"Orland Hoeber","is_ca":true},{"name":"Xue Yang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.009736650522841726,"gpt":0.2347955252822921,"spread":0.2250588747594504,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002759717,0.002378796,0.001404244,0.006489641,0.0006789435,0.003779009,0.001836573,0.001486763,0.04882411],"category_scores_gemma":[0.01391238,0.0008491825,0.0009503111,0.005457579,0.0008593377,0.007450326,0.003733596,0.0009414828,0.01643904],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007395268,"about_ca_system_score_gemma":0.0005539018,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001452547,"about_ca_topic_score_gemma":0.001745671,"domain_scores_codex":[0.9980093,0.0006568565,0.0003141832,0.000274495,0.0005921334,0.0001530652],"domain_scores_gemma":[0.9910113,0.006316732,0.000448395,0.0007238102,0.001180948,0.0003188551],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002685704,0.000311484,0.001755661,0.002965143,0.0001702708,0.001431553,0.003802317,0.00562846,0.04157156,0.03485169,0.1803419,0.7244843],"study_design_scores_gemma":[0.001042592,0.0007129612,0.002063445,0.001242835,0.0002161543,0.001885288,0.002083881,0.1152577,0.07189903,0.1080795,0.6950029,0.0005137692],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02008684,0.001715923,0.734092,0.0007599405,0.0004455325,0.000789423,0.005776451,0.2120498,0.02428404],"genre_scores_gemma":[0.1403288,0.001993184,0.7967836,0.0009168693,0.0003551313,0.001721655,0.01277805,0.01638605,0.02873668],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04882411,"threshold_uncertainty_score":0.1633329,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2166597489","doi":"10.1145/1978942.1979205","title":"Characterizing the usability of interactive applications through query log analysis","year":2011,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Usability; Computer science; Troubleshooting; Web search query; Set (abstract data type); Information retrieval; Selection (genetic algorithm); Web query classification; Process (computing); Product (mathematics); Usability inspection; The Internet; Query expansion; Heuristic evaluation; Search engine; World Wide Web; Human–computer interaction; Artificial intelligence","authors":[{"name":"Adam Fourney","is_ca":true},{"name":"Richard Mann","is_ca":true},{"name":"Michael Terry","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04529209062245151,"gpt":0.2802455875017747,"spread":0.2349534968793232,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01024026,0.0007988738,0.0007263896,0.007060168,0.0006863389,0.002555268,0.0008684641,0.0007423583,0.0004910577],"category_scores_gemma":[0.07077404,0.0003740913,0.0006093843,0.004492343,0.0009307053,0.003220123,0.001264935,0.001130668,0.000304923],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007815407,"about_ca_system_score_gemma":0.0007877504,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006333866,"about_ca_topic_score_gemma":0.008335097,"domain_scores_codex":[0.9869012,0.005952141,0.001477867,0.0009240329,0.004293564,0.0004511935],"domain_scores_gemma":[0.8460579,0.1237482,0.009525083,0.00998084,0.009809538,0.0008783499],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00158795,0.001676655,0.6130834,0.001307874,0.0005368385,0.0004445805,0.01080184,0.01976259,0.03026129,0.003274658,0.003500604,0.3137617],"study_design_scores_gemma":[0.00009622318,0.002523347,0.6692505,0.0001881543,0.0003149736,0.0007587814,0.006643532,0.2730031,0.03019226,0.01053112,0.006204184,0.0002938149],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9361542,0.0003426906,0.05925287,0.0001575205,0.00001041534,0.0003706956,0.001049013,0.0008973764,0.00176522],"genre_scores_gemma":[0.9540259,0.0001773314,0.04294873,0.00005239304,0.00001901581,0.0003782614,0.001891359,0.000106907,0.0004001097],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01024026,"threshold_uncertainty_score":0.0541563,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1972495172","doi":"10.1145/2723372.2723725","title":"TEGRA","year":2015,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Information retrieval; Row; Table (database); Relational database; Schema (genetic algorithms); World Wide Web; Data mining; Database","authors":[{"name":"Xu Chu","is_ca":true},{"name":"Yeye He","is_ca":false},{"name":"Kaushik Chakrabarti","is_ca":false},{"name":"Kris Ganjam","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05405638443133703,"gpt":0.2543020114080482,"spread":0.2002456269767112,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001715157,0.001221643,0.0008845153,0.002831692,0.001917685,0.006846135,0.002016395,0.002635883,0.4085029],"category_scores_gemma":[0.004577252,0.0006176491,0.0008104233,0.002159743,0.0009918772,0.003123365,0.004187234,0.002219223,0.3504872],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001880904,"about_ca_system_score_gemma":0.002702109,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00297971,"about_ca_topic_score_gemma":0.002546622,"domain_scores_codex":[0.9977955,0.0003572581,0.000164991,0.0005780408,0.0008276562,0.0002766396],"domain_scores_gemma":[0.9977963,0.0002814173,0.000188512,0.0005795885,0.0007303238,0.0004239087],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005938322,0.0001742162,0.002672923,0.0009541686,0.00006733648,0.000960056,0.0006753035,0.00082484,0.005611499,0.06674267,0.3776146,0.5431085],"study_design_scores_gemma":[0.00001837126,0.00003272097,0.0005480079,0.0001231382,0.00001336794,0.0004615318,0.0001044837,0.0003296171,0.001287921,0.003791531,0.9932758,0.00001361775],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.00766729,0.01274943,0.02400012,0.008833779,0.006582872,0.000380318,0.01071193,0.008037601,0.9210367],"genre_scores_gemma":[0.05747311,0.008074243,0.0170334,0.003759641,0.001680678,0.000318975,0.0121261,0.002410649,0.8971232],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.4085029,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W108479605","doi":"","title":"A New Similarity Measure to Understand Visitor Behavior in a Web Site","year":2004,"lang":"en","type":"article","venue":"IEICE Transactions on Information and Systems","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Hatch (Canada)","funders":"","keywords":"Visitor pattern; Computer science; World Wide Web; Cluster analysis; Similarity (geometry); Web page; Similarity measure; The Internet; Web site; Information retrieval; Measure (data warehouse); Web analytics; Site map; Order (exchange); Web mining; Web navigation; Web modeling; Static web page; Data mining; Web intelligence; Artificial intelligence","authors":[{"name":"Juan D. Velásquez","is_ca":false},{"name":"Hiroshi Yasuda","is_ca":false},{"name":"Terumasa Aoki","is_ca":true},{"name":"Richard W. Weber","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02053953015793934,"gpt":0.2451619656474609,"spread":0.2246224354895215,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001780091,0.0007964119,0.001478488,0.01113731,0.0008021494,0.001832666,0.001383285,0.001617096,0.001676736],"category_scores_gemma":[0.0109529,0.000221587,0.001138828,0.008219316,0.0009332364,0.004571724,0.001222754,0.00115275,0.0008176277],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001293362,"about_ca_system_score_gemma":0.000785426,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003256381,"about_ca_topic_score_gemma":0.00246294,"domain_scores_codex":[0.9972194,0.000502947,0.0004416666,0.0005113173,0.001134405,0.0001902374],"domain_scores_gemma":[0.9948318,0.002261561,0.0007980759,0.0006587204,0.001130232,0.0003195994],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008840754,0.001759741,0.1709771,0.001164905,0.001368504,0.0005543911,0.002693098,0.05887028,0.03981537,0.06152724,0.01414132,0.646244],"study_design_scores_gemma":[0.00006942982,0.001248419,0.1457387,0.0001891506,0.0003901052,0.00290682,0.001816901,0.7534676,0.01160432,0.06120458,0.02105971,0.0003043147],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2263987,0.002161468,0.7598217,0.0003363445,0.0001658744,0.0005457434,0.002216385,0.00136022,0.006993638],"genre_scores_gemma":[0.7512327,0.0004586921,0.2431449,0.0001165448,0.0001717365,0.0004159684,0.002475019,0.0001472623,0.001837065],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01113731,"threshold_uncertainty_score":0.009414136,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1533716782","doi":"10.1007/978-3-540-28637-0_20","title":"Web Page Transformation When Switching Devices","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Transformation (genetics); World Wide Web","authors":[{"name":"Bonnie MacKay","is_ca":true},{"name":"Carolyn Watters","is_ca":true},{"name":"Jack Duffy","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01583395283593206,"gpt":0.2348622141047844,"spread":0.2190282612688523,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006211714,0.0005239967,0.0005586455,0.001025545,0.0006106913,0.002519,0.001108905,0.001027434,0.02077595],"category_scores_gemma":[0.006421506,0.000486403,0.0005963956,0.0009777001,0.0006042462,0.003780731,0.001324575,0.001075527,0.004974987],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003028775,"about_ca_system_score_gemma":0.000457459,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001119179,"about_ca_topic_score_gemma":0.001366643,"domain_scores_codex":[0.9987663,0.000128616,0.00007860711,0.0002249821,0.0005445966,0.0002569824],"domain_scores_gemma":[0.9957494,0.001393285,0.0001564339,0.001963113,0.0005488801,0.0001888432],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002719008,0.0009640906,0.01027002,0.0004967659,0.00009007986,0.004103505,0.001242261,0.01478311,0.1126461,0.03896714,0.05682541,0.7568924],"study_design_scores_gemma":[0.0001950688,0.0005301654,0.01382974,0.0002236185,0.0002442358,0.003943853,0.002258652,0.3336196,0.4105013,0.1063935,0.1281184,0.0001420082],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"other","genre_scores_codex":[0.4161316,0.000709243,0.3888539,0.001108382,0.001638794,0.0006019096,0.001823985,0.06708189,0.1220503],"genre_scores_gemma":[0.8836876,0.0002986646,0.06884689,0.00027678,0.0001114208,0.0001066543,0.001944709,0.004391633,0.04033556],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.02077595,"threshold_uncertainty_score":0.06950247,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2039447541","doi":"10.1145/564376.564448","title":"The impact of corpus size on question answering performance","year":2002,"lang":"en","type":"article","venue":"","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Terabyte; Question answering; Computer science; Information retrieval; World Wide Web; Natural language processing","authors":[{"name":"Charles L. A. Clarke","is_ca":true},{"name":"Gordon V. Cormack","is_ca":true},{"name":"Michael Laszlo","is_ca":true},{"name":"Thomas R. Lynam","is_ca":true},{"name":"Egidio L. Terra","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01666226182163508,"gpt":0.2518497277000022,"spread":0.2351874658783671,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01604118,0.001266864,0.002080804,0.002302729,0.002031347,0.004490243,0.001954387,0.002123216,0.005413017],"category_scores_gemma":[0.1336818,0.001226227,0.0008176861,0.003299821,0.001506777,0.009438394,0.003028657,0.001842038,0.00316816],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001257331,"about_ca_system_score_gemma":0.001937355,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006996242,"about_ca_topic_score_gemma":0.007310003,"domain_scores_codex":[0.9848387,0.007107618,0.002057667,0.002479654,0.002858903,0.0006574888],"domain_scores_gemma":[0.8332128,0.142969,0.001974744,0.008808692,0.01106298,0.001971799],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01593118,0.002993579,0.05905309,0.004737078,0.001162855,0.001932565,0.004540185,0.0594523,0.1594008,0.004260576,0.0978894,0.5886463],"study_design_scores_gemma":[0.002837961,0.01055252,0.08905075,0.0007058725,0.002627199,0.004648715,0.004963093,0.5001194,0.3007023,0.0155989,0.06743614,0.000757053],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9362507,0.008690783,0.01787249,0.004187755,0.00106907,0.0004252201,0.005305243,0.01232332,0.01387537],"genre_scores_gemma":[0.9433951,0.002666305,0.02718621,0.0009258395,0.000705206,0.0006459939,0.01666844,0.002241198,0.005565571],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01604118,"threshold_uncertainty_score":0.08483487,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}