{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":329,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":329,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"416db036aa93","filters":{"venue":"Proceedings of the VLDB Endowment"}},"results":[{"id":"W2591700809","doi":"10.14778/3137628.3137631","title":"HoloClean","year":2017,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":454,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency","keywords":"Leverage (statistics); Computer science; Probabilistic logic; Inference; Tuple; Data mining; Statistical model; Machine learning; Artificial intelligence; Mathematics","authors":[{"name":"Theodoros Rekatsinas","is_ca":false},{"name":"Xu Chu","is_ca":true},{"name":"Ihab F. Ilyas","is_ca":true},{"name":"Christopher Ré","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2284211371904144,"gpt":0.4286142488382545,"spread":0.2001931116478401,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004578141,0.001504516,0.001491478,0.00361551,0.001016351,0.005033804,0.006821305,0.001534417,0.01789622],"category_scores_gemma":[0.02542172,0.001209874,0.003122862,0.00255591,0.001827585,0.006923518,0.007741638,0.003278982,0.00826311],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00213188,"about_ca_system_score_gemma":0.005024464,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008731898,"about_ca_topic_score_gemma":0.01582726,"domain_scores_codex":[0.9949383,0.0007989898,0.0003875704,0.001465794,0.002102678,0.0003067198],"domain_scores_gemma":[0.9922168,0.002418151,0.0004580509,0.003561347,0.001102045,0.0002435548],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005564808,0.0002067958,0.006176967,0.001574929,0.0003254147,0.0004643956,0.000695684,0.06786154,0.004960692,0.1164097,0.1236348,0.6771326],"study_design_scores_gemma":[0.0001088519,0.0001124345,0.001084319,0.0003362753,0.00009390851,0.0006494065,0.0001737318,0.4664161,0.01203458,0.2181794,0.3006937,0.0001172597],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002447318,0.0008144128,0.9284461,0.0006306557,0.0001884736,0.0002584502,0.003887755,0.05710077,0.006226082],"genre_scores_gemma":[0.0565011,0.0006951344,0.9088446,0.001032132,0.0001207292,0.0004492227,0.01686006,0.007929657,0.007567351],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01789622,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1991635064","doi":"10.14778/2047485.2047492","title":"A data-based approach to social influence maximization","year":2011,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Complex Network Analysis Techniques","field":"Physics and Astronomy","cited_by":426,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Maximization; Submodular set function; Computer science; Scalability; Perspective (graphical); Set (abstract data type); Function (biology); Expectation–maximization algorithm; Social graph; Social network (sociolinguistics); Mathematical optimization; Theoretical computer science; Artificial intelligence; Social media; Mathematics; Maximum likelihood; World Wide Web; Statistics","authors":[{"name":"Amit Goyal","is_ca":true},{"name":"Francesco Bonchi","is_ca":false},{"name":"Laks V. S. Lakshmanan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05672051885013572,"gpt":0.261899610639344,"spread":0.2051790917892083,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004309889,0.001920257,0.002628489,0.002680773,0.0009655249,0.002829244,0.004028468,0.002747652,0.002675511],"category_scores_gemma":[0.02209104,0.001104108,0.001850612,0.004322161,0.002263358,0.005101502,0.002967305,0.003887398,0.0007443896],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002906474,"about_ca_system_score_gemma":0.001599906,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003881742,"about_ca_topic_score_gemma":0.004190293,"domain_scores_codex":[0.9965501,0.001298967,0.0001898503,0.0009792532,0.0007990678,0.0001828044],"domain_scores_gemma":[0.9871398,0.009606346,0.0008665477,0.00118428,0.0009034892,0.00029952],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001296655,0.0001734243,0.003062385,0.000330525,0.0002030631,0.0001608367,0.0002573589,0.7798827,0.001581639,0.1365348,0.005212234,0.07247142],"study_design_scores_gemma":[0.0000112683,0.00001829161,0.0001474789,0.00001417097,0.0000129138,0.00003675374,0.00001659135,0.9315203,0.0005871168,0.0663912,0.001234878,0.000008946231],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004838125,0.0002647919,0.9925609,0.0004830493,0.00002867487,0.00009538738,0.000339318,0.0002029189,0.001186804],"genre_scores_gemma":[0.4564472,0.00117393,0.5344872,0.0004963736,0.0005444019,0.0008956415,0.001760273,0.0002489877,0.003946037],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004309889,"threshold_uncertainty_score":0.02279317,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2128248866","doi":"10.14778/1687627.1687734","title":"k-automorphism","year":2009,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":407,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Popularity; Computer science; Extension (predicate logic); Personally identifiable information; Automorphism; Computer security; Data mining; Theoretical computer science; Mathematics; Discrete mathematics; Programming language","authors":[{"name":"Lei Zou","is_ca":false},{"name":"Lei Chen","is_ca":false},{"name":"M. TAMER ÖZSU","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01853108214018638,"gpt":0.2408935676521065,"spread":0.2223624855119201,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003147309,0.001170624,0.001898382,0.001411583,0.00272334,0.00360982,0.001976487,0.002809147,0.006997627],"category_scores_gemma":[0.02057558,0.0008142953,0.003386925,0.001259379,0.005040418,0.008743367,0.006341694,0.00428636,0.005495713],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001488464,"about_ca_system_score_gemma":0.002218748,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006060633,"about_ca_topic_score_gemma":0.00041228,"domain_scores_codex":[0.9948801,0.001179463,0.0005402142,0.001645408,0.001006848,0.0007479775],"domain_scores_gemma":[0.9796474,0.007429112,0.001598389,0.008787775,0.001816739,0.0007204728],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000630856,0.0001941259,0.003641982,0.0005633452,0.0002342329,0.000547422,0.001684479,0.03075872,0.01244976,0.8233978,0.01812693,0.1077703],"study_design_scores_gemma":[0.00006793401,0.0001734329,0.0005031736,0.00005842104,0.00006008905,0.0008628143,0.0002309194,0.04515909,0.01015387,0.9264411,0.01622251,0.00006680098],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.08529949,0.000643977,0.8767,0.002138962,0.0003918103,0.0003943824,0.0008312298,0.003171392,0.03042864],"genre_scores_gemma":[0.7756864,0.0008624443,0.200435,0.001257392,0.0003982674,0.0003762157,0.00130225,0.000841552,0.01884047],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006997627,"threshold_uncertainty_score":0.02340937,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1967167578","doi":"10.14778/1453856.1453980","title":"Discovering data quality rules","year":2008,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":268,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Data quality; Consistency (knowledge bases); Context (archaeology); Quality (philosophy); Data mining; Data integrity; Set (abstract data type); Data consistency; Process (computing); Database; Artificial intelligence; Programming language; Engineering","authors":[{"name":"Fei Chiang","is_ca":true},{"name":"Renée J. Miller","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4977714110126681,"gpt":0.451736219472416,"spread":0.04603519154025215,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01511977,0.001813347,0.002101456,0.01194249,0.001823355,0.007347757,0.004192876,0.002377663,0.00227075],"category_scores_gemma":[0.10615,0.001572119,0.003998329,0.005548572,0.001834089,0.008104751,0.00468046,0.003941247,0.001131723],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002808427,"about_ca_system_score_gemma":0.006522246,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008356454,"about_ca_topic_score_gemma":0.01122686,"domain_scores_codex":[0.9650422,0.004980867,0.005216945,0.005815096,0.01771588,0.001228974],"domain_scores_gemma":[0.8815446,0.07096115,0.009725274,0.01473743,0.02121326,0.001818291],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005732684,0.0009830564,0.1674478,0.003636991,0.001030492,0.003484358,0.002749587,0.05252321,0.01577806,0.06386509,0.03590264,0.6520255],"study_design_scores_gemma":[0.0001905392,0.0003615384,0.02417933,0.001836159,0.0007876817,0.002844309,0.002675211,0.6406669,0.0491715,0.1506765,0.1263214,0.0002889383],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09791879,0.002485348,0.8449972,0.005924887,0.0002697595,0.002182289,0.02652693,0.0113244,0.008370271],"genre_scores_gemma":[0.213557,0.0008566868,0.7521936,0.0009519063,0.0001054793,0.0007087578,0.02934587,0.0007992009,0.0014815],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01511977,"threshold_uncertainty_score":0.0799619,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1982177147","doi":"10.14778/2002974.2002976","title":"gStore","year":2011,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":263,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"SPARQL; RDF; Computer science; Information retrieval; Named graph; RDF Schema; Linked data; RDF query language; Scalability; RDF/XML; Pruning; Database; Semantic Web; Web search query; Web query classification; Search engine","authors":[{"name":"Lei Zou","is_ca":false},{"name":"Jinghui Mo","is_ca":false},{"name":"Lei Chen","is_ca":false},{"name":"M. TAMER ÖZSU","is_ca":true},{"name":"Dongyan Zhao","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03354416418142819,"gpt":0.1981530133423381,"spread":0.1646088491609099,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007645193,0.001325793,0.001223977,0.001565456,0.00114446,0.003443604,0.002953996,0.001502919,0.230638],"category_scores_gemma":[0.00255073,0.000959144,0.001425723,0.002620814,0.0004822961,0.004507332,0.003397306,0.001973308,0.1847206],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009796979,"about_ca_system_score_gemma":0.001526402,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003069328,"about_ca_topic_score_gemma":0.004725919,"domain_scores_codex":[0.9988155,0.0001357717,0.00009864494,0.0003431845,0.0004316955,0.0001751832],"domain_scores_gemma":[0.9989249,0.0001405917,0.00005255286,0.0005193547,0.0002620418,0.0001004768],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004497201,0.000150872,0.001271919,0.0007304072,0.0001186959,0.000247829,0.0001621465,0.002182583,0.007022507,0.0234921,0.7279258,0.2362453],"study_design_scores_gemma":[0.0001216316,0.00006012377,0.0006603679,0.00006039295,0.00003467443,0.0003059394,0.00007035922,0.01151903,0.008486561,0.01797107,0.9606709,0.00003896318],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.008428562,0.002466371,0.2025956,0.002551394,0.002078335,0.0006671418,0.06884974,0.3455576,0.3668053],"genre_scores_gemma":[0.07981606,0.002431221,0.2034926,0.003769108,0.0005277935,0.0009835127,0.301625,0.05060912,0.3567455],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.230638,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2170712852","doi":"10.14778/2536258.2536262","title":"Discovering denial constraints","year":2013,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":252,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Rotation formalisms in three dimensions; Scalability; Inference; Data integrity; Set (abstract data type); Functional dependency; Constraint (computer-aided design); Theoretical computer science; Rank (graph theory); Semantics (computer science); Function (biology); Data mining; Artificial intelligence; Programming language; Database; Relational database","authors":[{"name":"Xu Chu","is_ca":true},{"name":"Ihab F. Ilyas","is_ca":false},{"name":"Paolo Papotti","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08220481571088507,"gpt":0.3393964871060694,"spread":0.2571916713951843,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01002849,0.001878893,0.002140607,0.005793695,0.00281759,0.005422847,0.004843651,0.002492945,0.006076572],"category_scores_gemma":[0.06524177,0.001217365,0.002575205,0.00521947,0.001869019,0.01194757,0.005861333,0.004345814,0.001468525],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002752434,"about_ca_system_score_gemma":0.006749071,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008869141,"about_ca_topic_score_gemma":0.01100694,"domain_scores_codex":[0.9747038,0.007525744,0.002251216,0.00443259,0.009265797,0.001820832],"domain_scores_gemma":[0.9453601,0.03358305,0.00326691,0.008111537,0.008476475,0.001201925],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000666251,0.0005313721,0.03876958,0.001719955,0.0005265963,0.001871795,0.001223383,0.1119818,0.0111726,0.2364928,0.05681894,0.5382251],"study_design_scores_gemma":[0.00007268655,0.00007926219,0.002910353,0.0002359403,0.0001420032,0.001165007,0.001145249,0.6139483,0.0162853,0.3189627,0.04493577,0.0001174735],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06368043,0.001013549,0.9088178,0.004080155,0.0002533149,0.0008495224,0.006414272,0.003734162,0.01115685],"genre_scores_gemma":[0.3549635,0.000549057,0.6268752,0.001354953,0.000182421,0.0004705918,0.01107964,0.0007399461,0.003784626],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01002849,"threshold_uncertainty_score":0.05303633,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2406955896","doi":"10.14778/2732219.2732227","title":"Multi-core, main-memory joins","year":2013,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Advanced Data Storage Technologies","field":"Computer Science","cited_by":247,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Computer science; Hash join; Joins; Merge sort; Parallel computing; Join (topology); sort; Merge (version control); Hash function; Merge algorithm; SIMD; Theoretical computer science; Sorting algorithm; Database; Programming language; Mathematics","authors":[{"name":"Çagri Balkesen","is_ca":false},{"name":"Gustavo Alonso","is_ca":false},{"name":"Jens Teubner","is_ca":false},{"name":"M. TAMER ÖZSU","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02508474588165716,"gpt":0.2397200588904825,"spread":0.2146353130088254,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002464928,0.0006757801,0.0007613237,0.0007369681,0.001114033,0.001389011,0.002129611,0.0006953703,0.003863802],"category_scores_gemma":[0.007439917,0.0004097286,0.0003272672,0.001297137,0.0006510011,0.003355035,0.001408781,0.0008502881,0.0008797198],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007112357,"about_ca_system_score_gemma":0.001251651,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009041899,"about_ca_topic_score_gemma":0.00130049,"domain_scores_codex":[0.9968611,0.0002696138,0.0002216882,0.0006484922,0.001610381,0.0003887174],"domain_scores_gemma":[0.9932334,0.002767608,0.0004752045,0.001759018,0.001420929,0.0003438618],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.009230415,0.003596002,0.0221136,0.001612835,0.000411631,0.0006497273,0.00131878,0.1957367,0.2969776,0.02690451,0.02352651,0.4179217],"study_design_scores_gemma":[0.0003800692,0.003375055,0.006190443,0.00004844721,0.0001288296,0.0007282224,0.0006006904,0.5613014,0.3936812,0.01195358,0.02153931,0.00007278998],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8507018,0.002085744,0.1216985,0.0002428837,0.0003347499,0.0003188787,0.0006706389,0.005470775,0.01847602],"genre_scores_gemma":[0.90058,0.0003400831,0.09283619,0.0001824361,0.00005852857,0.0001608709,0.000928259,0.0004885525,0.004425097],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003863802,"threshold_uncertainty_score":0.01303595,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2548122763","doi":"10.14778/2994509.2994514","title":"ActiveClean","year":2016,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":244,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"MNIST database; Computer science; Context (archaeology); Support vector machine; Data mining; Convergence (economics); Process (computing); Class (philosophy); Iterative and incremental development; Machine learning; Artificial intelligence; Deep learning","authors":[{"name":"Sanjay Krishnan","is_ca":false},{"name":"Jiannan Wang","is_ca":true},{"name":"Eugene Wu","is_ca":false},{"name":"Michael J. Franklin","is_ca":false},{"name":"Ken Goldberg","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.009115700324434017,"gpt":0.2121947867126384,"spread":0.2030790863882044,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005524416,0.003281085,0.002910606,0.003398271,0.002098419,0.007996758,0.007369641,0.003558987,0.04162381],"category_scores_gemma":[0.02294586,0.002480949,0.003860537,0.003341832,0.001404523,0.007642557,0.007066833,0.00456045,0.03641694],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008763244,"about_ca_system_score_gemma":0.003094282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003690959,"about_ca_topic_score_gemma":0.007741269,"domain_scores_codex":[0.9951501,0.001209463,0.0004006198,0.001279485,0.001660123,0.0003001515],"domain_scores_gemma":[0.990415,0.004144664,0.0003539688,0.003346487,0.0014871,0.0002528988],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005766489,0.0002924403,0.004378524,0.001855119,0.0006870286,0.0004468172,0.0008711849,0.05785158,0.006938002,0.04163717,0.4085335,0.475932],"study_design_scores_gemma":[0.0001804244,0.0001143024,0.0008681212,0.0002539227,0.0001296521,0.0006816341,0.0004368314,0.435555,0.01938186,0.1141362,0.4281086,0.0001533892],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00346497,0.001361244,0.8787732,0.001261081,0.000674281,0.0003527406,0.006805339,0.09451578,0.01279131],"genre_scores_gemma":[0.06609979,0.002148384,0.8232331,0.002652606,0.0003656152,0.001181668,0.04470693,0.03090976,0.02870207],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04162381,"threshold_uncertainty_score":0.1392455,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2621145626","doi":"10.14778/3099622.3099626","title":"Attribute-driven community search","year":2017,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Complex Network Analysis Techniques","field":"Physics and Astronomy","cited_by":241,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Node (physics); Theoretical computer science; Relevance (law); Graph; Community structure; Cohesion (chemistry); Data mining; Mathematics; Combinatorics","authors":[{"name":"Xin Huang","is_ca":false},{"name":"Laks V. S. Lakshmanan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04102465157361466,"gpt":0.3032241378313941,"spread":0.2621994862577794,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003737529,0.001184398,0.002556851,0.004954778,0.001829028,0.002030328,0.003495515,0.002509891,0.002939781],"category_scores_gemma":[0.01403862,0.0007495718,0.001611041,0.007300162,0.001404411,0.004832463,0.003525081,0.001592025,0.0009970963],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001759728,"about_ca_system_score_gemma":0.001706588,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004812211,"about_ca_topic_score_gemma":0.005608539,"domain_scores_codex":[0.996046,0.001433999,0.0001904343,0.001007239,0.00108885,0.0002334505],"domain_scores_gemma":[0.9917973,0.004938946,0.0006822974,0.0009126957,0.001280363,0.000388345],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004235626,0.0004196332,0.005092202,0.0008740696,0.0005221784,0.0002714834,0.0007636474,0.5084456,0.003964516,0.1194162,0.01959518,0.3402117],"study_design_scores_gemma":[0.00007111157,0.00004869896,0.0002460058,0.00002405989,0.00003935279,0.0001148669,0.00007082218,0.9058746,0.0007551567,0.08918751,0.003549638,0.00001816434],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02493704,0.001811277,0.9679136,0.0007425855,0.00009328967,0.0003125884,0.000624452,0.0006334327,0.002931734],"genre_scores_gemma":[0.4121432,0.001487394,0.5760769,0.0007970922,0.0003259825,0.000573586,0.002955321,0.0002162165,0.005424307],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004954778,"threshold_uncertainty_score":0.01976615,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2544486974","doi":"10.14778/2994509.2994518","title":"Detecting data errors","year":2016,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":237,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"University of California Berkeley","keywords":"Computer science; Raw data; Outlier; Data mining; Set (abstract data type); Variety (cybernetics); Data quality; Ground truth; Anomaly detection; Quality (philosophy); Big data; Data science; Machine learning; Artificial intelligence; Engineering","authors":[{"name":"Ziawasch Abedjan","is_ca":false},{"name":"Xu Chu","is_ca":true},{"name":"Dong Deng","is_ca":false},{"name":"Raul Castro Fernandez","is_ca":false},{"name":"Ihab F. Ilyas","is_ca":true},{"name":"Mourad Ouzzani","is_ca":false},{"name":"Paolo Papotti","is_ca":false},{"name":"Michael Stonebraker","is_ca":false},{"name":"Nan Tang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3097609978046282,"gpt":0.4126229492891771,"spread":0.1028619514845489,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02098097,0.002211681,0.001998074,0.009275529,0.001645167,0.005104362,0.004321513,0.002830164,0.002670607],"category_scores_gemma":[0.1517903,0.000865113,0.002016069,0.007719268,0.001597924,0.005799147,0.006564772,0.002727047,0.002860892],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001315207,"about_ca_system_score_gemma":0.003249373,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002358616,"about_ca_topic_score_gemma":0.002255817,"domain_scores_codex":[0.9522556,0.009320097,0.007339614,0.01126091,0.01824003,0.00158373],"domain_scores_gemma":[0.7991547,0.07963275,0.01970252,0.06453487,0.03549375,0.001481378],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009512166,0.0005889611,0.1189586,0.003738887,0.0007083282,0.002322092,0.008237539,0.01397714,0.03586598,0.02107788,0.04859595,0.7449773],"study_design_scores_gemma":[0.0002018019,0.0008980662,0.07099519,0.002114834,0.0008785389,0.004792019,0.008833255,0.21652,0.3119785,0.07103218,0.3110456,0.000709948],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.117291,0.002201726,0.810873,0.003246109,0.0008943844,0.001943357,0.01510004,0.03976662,0.008683854],"genre_scores_gemma":[0.2672631,0.0006566591,0.7069734,0.001401538,0.0001439539,0.0009840429,0.01583991,0.002757099,0.003980435],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02098097,"threshold_uncertainty_score":0.1109593,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2970992672","doi":"10.14778/3352063.3352116","title":"Data lake management","year":2019,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":236,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"TD Bank Group; University of Toronto","funders":"","keywords":"Metadata; Data management; Metadata management; Data science; Data management plan; Computer science; Data integration; Software versioning; Data mapping; Data extraction; Data element; Research data; Data virtualization; Data curation; Database; World Wide Web; Software","authors":[{"name":"Fatemeh Nargesian","is_ca":true},{"name":"Erkang Zhu","is_ca":true},{"name":"Renée J. Miller","is_ca":false},{"name":"Ken Q. Pu","is_ca":false},{"name":"Patricia C. Arocena","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.190965546163535,"gpt":0.3896089267691528,"spread":0.1986433806056178,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01351263,0.001064766,0.001422133,0.007892552,0.003122945,0.01269661,0.005567789,0.00149881,0.0280816],"category_scores_gemma":[0.0414289,0.001078477,0.001497977,0.009107835,0.001126833,0.01718519,0.01127162,0.00326425,0.01830473],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002542973,"about_ca_system_score_gemma":0.007144351,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006534911,"about_ca_topic_score_gemma":0.004540147,"domain_scores_codex":[0.9902188,0.001388216,0.001680583,0.001443264,0.004585288,0.0006839008],"domain_scores_gemma":[0.9779451,0.003213963,0.001534096,0.009015027,0.006640497,0.001651296],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002141557,0.00008546647,0.006912315,0.0009170364,0.0001310364,0.0002502174,0.00113894,0.002427701,0.003093446,0.07858947,0.5607018,0.3455383],"study_design_scores_gemma":[0.00003398857,0.00003280322,0.001850776,0.0002551512,0.00004663851,0.0002070739,0.0004127493,0.01122277,0.005973069,0.0329357,0.9469594,0.00006990021],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01169187,0.007111954,0.5594096,0.02324033,0.002832851,0.003080809,0.1010732,0.1775492,0.1140102],"genre_scores_gemma":[0.1118763,0.007822782,0.5482424,0.005196006,0.001630319,0.002458069,0.222079,0.02403266,0.07666242],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0280816,"threshold_uncertainty_score":0.09394228,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2148524305","doi":"10.14778/1687627.1687771","title":"Framework for evaluating clustering algorithms in duplicate detection","year":2009,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":232,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Scalability; Cluster analysis; Data mining; Data deduplication; Process (computing); Algorithm; Machine learning; Database","authors":[{"name":"Oktie Hassanzadeh","is_ca":true},{"name":"Fei Chiang","is_ca":true},{"name":"Hyun‐Chul Lee","is_ca":false},{"name":"Renée J. Miller","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2418184006816228,"gpt":0.4629068859449667,"spread":0.2210884852633439,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05879213,0.002601526,0.002919211,0.01180054,0.002354739,0.005690672,0.005376305,0.00425544,0.001794421],"category_scores_gemma":[0.1066421,0.0008347164,0.002408877,0.009012589,0.002695965,0.00483065,0.005912534,0.002610947,0.0008506694],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004568572,"about_ca_system_score_gemma":0.004954954,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007920526,"about_ca_topic_score_gemma":0.005157467,"domain_scores_codex":[0.9419968,0.03202862,0.004135181,0.003337024,0.01725783,0.001244553],"domain_scores_gemma":[0.9364167,0.03646897,0.005568864,0.008679273,0.01173811,0.001128086],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009888812,0.001387541,0.01325356,0.001273341,0.001211442,0.0002533754,0.0007350098,0.5713453,0.009797089,0.1509697,0.008617209,0.2401675],"study_design_scores_gemma":[0.0001858201,0.001166475,0.002772674,0.0001680039,0.0001923019,0.0002233082,0.0002883164,0.9321498,0.005740625,0.04993129,0.007071983,0.0001093278],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02810548,0.001398938,0.9589446,0.0006685856,0.0001231674,0.00202075,0.0008784202,0.001899018,0.005961181],"genre_scores_gemma":[0.1118798,0.000395148,0.8841136,0.0001743555,0.00007874138,0.001652139,0.0008694024,0.0001627496,0.000674109],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.05879213,"threshold_uncertainty_score":0.3109262,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2131164945","doi":"10.14778/2367502.2367512","title":"Solving big data challenges for enterprise application performance management","year":2012,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Cloud Computing and Resource Management","field":"Computer Science","cited_by":232,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Scalability; Big data; Data science; Context (archaeology); Analytics; Instrumentation (computer programming); Data management; Enterprise system; System monitoring; Database; Data mining; Operating system","authors":[{"name":"Tilmann Rabl","is_ca":true},{"name":"Sergio Gómez-Villamor","is_ca":false},{"name":"Mohammad Sadoghi","is_ca":true},{"name":"Víctor Muntés-Mulero","is_ca":false},{"name":"Hans‐Arno Jacobsen","is_ca":true},{"name":"Serge Mankovskii","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0504586435494311,"gpt":0.2445988120777852,"spread":0.1941401685283541,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01855082,0.001729405,0.001917854,0.002790505,0.003008082,0.01515066,0.005218593,0.002720555,0.002257279],"category_scores_gemma":[0.0489126,0.001197821,0.00110552,0.006283352,0.002475858,0.02292174,0.007411846,0.008974658,0.001970385],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002282658,"about_ca_system_score_gemma":0.005188588,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005165065,"about_ca_topic_score_gemma":0.005510787,"domain_scores_codex":[0.9864094,0.003418882,0.001035796,0.001697846,0.00647645,0.0009616144],"domain_scores_gemma":[0.9498664,0.01862667,0.002708607,0.01262678,0.01192515,0.004246465],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00107554,0.0009508397,0.04943461,0.002124042,0.0006744353,0.001065043,0.006467273,0.06227835,0.01505208,0.09677759,0.1772706,0.5868296],"study_design_scores_gemma":[0.0001789476,0.0003936414,0.02493206,0.001097801,0.00022203,0.0009178726,0.01429171,0.3218508,0.0137893,0.3655129,0.2563973,0.0004157155],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1913507,0.03605011,0.4814318,0.2073299,0.005088327,0.001197044,0.008345593,0.0253377,0.04386874],"genre_scores_gemma":[0.6577726,0.01431348,0.2957341,0.008292442,0.004446295,0.0006322386,0.01124371,0.003052976,0.004512189],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01855082,"threshold_uncertainty_score":0.09810728,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2126547925","doi":"10.14778/1920841.1920906","title":"MRShare","year":2010,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Cloud Computing and Resource Management","field":"Computer Science","cited_by":226,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Cloud computing; Batch processing; Context (archaeology); Distributed computing; Work (physics); Core (optical fiber); Database; Operating system","authors":[{"name":"Tomasz Nykiel","is_ca":true},{"name":"Michalis Potamias","is_ca":false},{"name":"Chaitanya Mishra","is_ca":false},{"name":"George Kollios","is_ca":false},{"name":"Nick Koudas","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.00697231335455234,"gpt":0.1997996891760867,"spread":0.1928273758215344,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001538274,0.001422012,0.001249679,0.001472563,0.001077668,0.003805078,0.004228997,0.001612343,0.1842964],"category_scores_gemma":[0.005464657,0.0009656087,0.001448472,0.001480389,0.0006136687,0.005110882,0.004297202,0.002244269,0.1291546],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007288457,"about_ca_system_score_gemma":0.001807102,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001554403,"about_ca_topic_score_gemma":0.001941871,"domain_scores_codex":[0.9977192,0.000262201,0.0001454921,0.0004617169,0.001067982,0.0003434217],"domain_scores_gemma":[0.9971349,0.0003954714,0.0001200186,0.001215376,0.0007528387,0.0003813531],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009334768,0.0002416345,0.0009044626,0.0005906096,0.0001007108,0.0003546303,0.0001759829,0.004016144,0.008220611,0.02317191,0.762409,0.1988809],"study_design_scores_gemma":[0.0002161096,0.0001478319,0.0006361714,0.00005456369,0.00003120265,0.0004955349,0.00008354556,0.01951059,0.009782199,0.01647472,0.9524695,0.0000981141],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"software","genre_gemma":"empirical","genre_scores_codex":[0.01054719,0.001496139,0.2700234,0.003366326,0.002023326,0.001281939,0.03351859,0.4028381,0.2749048],"genre_scores_gemma":[0.153546,0.001905404,0.2534062,0.003094911,0.0009320015,0.001427603,0.1237304,0.05264042,0.4093171],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1842964,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2293703278","doi":"10.14778/3402707.3402744","title":"Publishing set-valued data via differential privacy","year":2011,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":221,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Concordia University","funders":"","keywords":"Differential privacy; Computer science; Data publishing; Data mining; Scalability; Context (archaeology); Data anonymization; Set (abstract data type); Information privacy; Information retrieval; Theoretical computer science; Publishing; Database; Computer security","authors":[{"name":"Rui Chen","is_ca":true},{"name":"Noman Mohammed","is_ca":true},{"name":"Benjamin C. M. Fung","is_ca":true},{"name":"Bipin C. Desai","is_ca":true},{"name":"Li Xiong","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09995646936477759,"gpt":0.2744083175834562,"spread":0.1744518482186786,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01635941,0.0007787725,0.002453239,0.001407328,0.001844657,0.006103734,0.003886112,0.002619027,0.001536135],"category_scores_gemma":[0.05496402,0.0009338431,0.001874068,0.004947859,0.004266058,0.01548024,0.007058446,0.004732266,0.0008515019],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001840026,"about_ca_system_score_gemma":0.002164657,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003655682,"about_ca_topic_score_gemma":0.0002503423,"domain_scores_codex":[0.9765705,0.009680839,0.001887711,0.003495441,0.007444179,0.0009213937],"domain_scores_gemma":[0.9385475,0.0295461,0.003474126,0.02532156,0.002262777,0.0008479986],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0009702999,0.0002914418,0.00542563,0.0004696366,0.0002731034,0.0009460576,0.001667178,0.0920082,0.01315459,0.673807,0.004744612,0.2062422],"study_design_scores_gemma":[0.000105851,0.0002602597,0.000715318,0.0000575076,0.00008462782,0.001393065,0.0003065754,0.3033261,0.01633059,0.6677849,0.009573798,0.00006153259],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0214358,0.0003557164,0.9740294,0.001571689,0.00005608634,0.0001286091,0.0003179726,0.0002476938,0.001857171],"genre_scores_gemma":[0.7013345,0.00107133,0.2911396,0.001005172,0.0003062083,0.0005074717,0.0008276575,0.0001365283,0.003671397],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01635941,"threshold_uncertainty_score":0.08651793,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2212315060","doi":"10.14778/2856318.2856323","title":"Approximate closest community search in networks","year":2015,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Caching and Content Delivery","field":"Computer Science","cited_by":217,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Truss; Computer science; Approximation algorithm; Greedy algorithm; Set (abstract data type); Graph; Efficient algorithm; Theoretical computer science; Mathematical optimization; Mathematics; Combinatorics; Algorithm","authors":[{"name":"Xin Huang","is_ca":true},{"name":"Laks V. S. Lakshmanan","is_ca":true},{"name":"Jeffrey Xu Yu","is_ca":false},{"name":"Hong Cheng","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05422769107867213,"gpt":0.2469106196643252,"spread":0.1926829285856531,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002277179,0.001632734,0.002631785,0.002651594,0.001548099,0.002038555,0.003694601,0.003006986,0.004041819],"category_scores_gemma":[0.01515344,0.001009356,0.001387598,0.00355746,0.001528871,0.005133215,0.00309819,0.002020596,0.0008805576],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002900251,"about_ca_system_score_gemma":0.001637695,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009859422,"about_ca_topic_score_gemma":0.0129611,"domain_scores_codex":[0.9978266,0.0007739355,0.0000871337,0.0006340445,0.0004457467,0.0002324201],"domain_scores_gemma":[0.9927394,0.005045403,0.0007005463,0.0005540361,0.0006320741,0.0003284829],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002074515,0.0001029705,0.00124389,0.0002661854,0.000104428,0.0001527625,0.0002055944,0.9099417,0.0009749124,0.03853371,0.005671347,0.042595],"study_design_scores_gemma":[0.00002247332,0.00001455796,0.00008391026,0.00001489223,0.000008394672,0.00003578314,0.0000373017,0.9634996,0.0001565914,0.03526317,0.0008574291,0.000005932113],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05239649,0.002555631,0.9357719,0.001224388,0.0001013044,0.0002249291,0.0006012628,0.0007531844,0.006370844],"genre_scores_gemma":[0.4901617,0.001561806,0.4968547,0.0004213523,0.0001655839,0.000481647,0.002007409,0.0003089631,0.008036776],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009859422,"threshold_uncertainty_score":0.02104294,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W808055529","doi":"10.14778/2850578.2850581","title":"From competition to complementarity","year":2015,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Complex Network Analysis Techniques","field":"Physics and Astronomy","cited_by":216,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Complementarity (molecular biology); Computer science; Maximization; Competition (biology); Mathematical optimization; Cellular automaton; Set (abstract data type); Margin (machine learning); Theoretical computer science; Artificial intelligence; Machine learning; Mathematics","authors":[{"name":"Wei Lu","is_ca":true},{"name":"Wei Chen","is_ca":false},{"name":"Laks V. S. Lakshmanan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0274108720421482,"gpt":0.2739003839945476,"spread":0.2464895119523994,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001652073,0.001436836,0.0015955,0.0009281597,0.001009876,0.001978939,0.001615521,0.00179077,0.005836039],"category_scores_gemma":[0.008870943,0.000811236,0.00158204,0.001208949,0.002117943,0.003622288,0.002514468,0.00225885,0.0004858085],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002069453,"about_ca_system_score_gemma":0.001384647,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005331788,"about_ca_topic_score_gemma":0.004615034,"domain_scores_codex":[0.9984126,0.0005992377,0.00004951936,0.0004785834,0.0002899018,0.0001701117],"domain_scores_gemma":[0.9947727,0.004141596,0.0003329836,0.0002759018,0.0002215578,0.0002552828],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001321359,0.0001268317,0.001914238,0.000262118,0.000119748,0.0003543972,0.0002337441,0.616694,0.001668656,0.3374773,0.007088011,0.03392882],"study_design_scores_gemma":[0.00002399581,0.00003894291,0.0003045998,0.00001812491,0.00002858051,0.0000916241,0.000044695,0.7991658,0.0006436449,0.1976176,0.002007058,0.00001533358],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0673905,0.0006615175,0.9090972,0.001788721,0.00005563816,0.0001597781,0.0003695749,0.0003597548,0.02011736],"genre_scores_gemma":[0.8588893,0.0007544078,0.1300892,0.0006618105,0.0001762296,0.0003762351,0.0005057031,0.0002129608,0.008334192],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005836039,"threshold_uncertainty_score":0.0195235,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2291620117","doi":"10.14778/2850469.2850471","title":"K-core decomposition of large networks on a single PC","year":2015,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Graph Theory and Algorithms","field":"Computer Science","cited_by":214,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Decomposition; Implementation; Core (optical fiber); Metric (unit); Graph; Multi-core processor; Vertex (graph theory); Theoretical computer science; Parallel computing; Algorithm","authors":[{"name":"Wissam Khaouid","is_ca":true},{"name":"Marina Barsky","is_ca":true},{"name":"Venkatesh Srinivasan","is_ca":true},{"name":"Alex Thomo","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03872705365176489,"gpt":0.2674381582051575,"spread":0.2287111045533927,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001044933,0.0008093466,0.000688081,0.0009295138,0.001004282,0.001489421,0.001622701,0.0007513224,0.004907934],"category_scores_gemma":[0.006103801,0.0004983993,0.000692374,0.001494264,0.000873805,0.00350875,0.001758532,0.001221925,0.001373386],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001557698,"about_ca_system_score_gemma":0.001722905,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006659556,"about_ca_topic_score_gemma":0.01219268,"domain_scores_codex":[0.9992757,0.0001394419,0.00004479252,0.000194746,0.0002080358,0.0001373329],"domain_scores_gemma":[0.9969958,0.00091551,0.0001971889,0.001301556,0.0004269131,0.000163088],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009009505,0.0003024004,0.005754163,0.0003796038,0.0001456611,0.0003656886,0.0007113432,0.6121204,0.01832686,0.06540336,0.01755828,0.2780312],"study_design_scores_gemma":[0.00003876512,0.00005571471,0.00058392,0.00001401064,0.00001610355,0.0000623538,0.0001594962,0.9538427,0.004528257,0.03723431,0.003454772,0.000009485914],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1806637,0.0003358126,0.8000292,0.0005712329,0.0000842087,0.0001789682,0.0004402823,0.006468044,0.01122853],"genre_scores_gemma":[0.5085057,0.000204562,0.4843883,0.000176829,0.00002603422,0.0002135139,0.001068089,0.0006452482,0.004771732],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006659556,"threshold_uncertainty_score":0.0164187,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2111607365","doi":"10.14778/1687627.1687727","title":"Distance-join","year":2009,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Graph Theory and Algorithms","field":"Computer Science","cited_by":207,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Reachability; Join (topology); Graph; Theoretical computer science; Query optimization; Shortest path problem; Data mining; Mathematics; Combinatorics","authors":[{"name":"Lei Zou","is_ca":false},{"name":"Lei Chen","is_ca":false},{"name":"M. TAMER ÖZSU","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.006708040847259433,"gpt":0.2017328283834329,"spread":0.1950247875361735,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002386493,0.001283068,0.002016978,0.003012727,0.001931449,0.003174458,0.004250894,0.001579977,0.02229116],"category_scores_gemma":[0.008770889,0.000670936,0.001654,0.005367994,0.0006916338,0.005055226,0.004442924,0.001840227,0.008290081],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009400303,"about_ca_system_score_gemma":0.001435233,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002195672,"about_ca_topic_score_gemma":0.003737273,"domain_scores_codex":[0.9952506,0.0006800601,0.0005134137,0.001286386,0.001969492,0.0003001595],"domain_scores_gemma":[0.9952947,0.001164407,0.0003038366,0.002207682,0.0007572607,0.0002720557],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001196244,0.0006818881,0.0036315,0.0008906203,0.0003523468,0.0002803667,0.0004701106,0.04489188,0.01064876,0.08609165,0.08818434,0.7626804],"study_design_scores_gemma":[0.0003860822,0.0007611435,0.001998132,0.0001244409,0.0002042013,0.001628282,0.0007702838,0.4901126,0.03217194,0.2459951,0.2256812,0.0001665319],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01479942,0.001453367,0.9535672,0.0006500403,0.0004368774,0.0006011936,0.004519517,0.008758062,0.01521421],"genre_scores_gemma":[0.1477823,0.0008740947,0.8205541,0.0004753306,0.0003240766,0.0005161673,0.01379753,0.00144472,0.01423155],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02229116,"threshold_uncertainty_score":0.07457131,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2294111665","doi":"10.14778/2732951.2732960","title":"Scalable logging through emerging non-volatile memory","year":2014,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Advanced Memory and Neural Computing","field":"Engineering","cited_by":206,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Commit; Scalability; Computer science; Logging; Dram; Bottleneck; Embedded system; Overhead (engineering); Cache; Operating system; Computer hardware; Database; Forestry","authors":[{"name":"Tianzheng Wang","is_ca":true},{"name":"Ryan Johnson","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.009201764651845435,"gpt":0.2140710501752543,"spread":0.2048692855234089,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007784016,0.0003227351,0.0003283535,0.0005071205,0.0006116311,0.001529041,0.002246613,0.0003952077,0.001994005],"category_scores_gemma":[0.003022725,0.0002968167,0.0001674779,0.0006430881,0.0006710574,0.00361997,0.002077036,0.001012736,0.0004741281],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005312896,"about_ca_system_score_gemma":0.001079503,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001402331,"about_ca_topic_score_gemma":0.00273687,"domain_scores_codex":[0.9993467,0.0001076971,0.00004837125,0.0001036932,0.0002876037,0.0001059793],"domain_scores_gemma":[0.9976764,0.000545478,0.0001778194,0.0009095019,0.0005612258,0.0001294957],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001729119,0.0006432558,0.01876345,0.0008644701,0.0001263472,0.00132883,0.001500287,0.1114268,0.1339176,0.07285539,0.0346894,0.6221551],"study_design_scores_gemma":[0.0002057588,0.0005803013,0.002554607,0.00010763,0.00007736764,0.0006991707,0.0009901886,0.778019,0.1248477,0.06097034,0.03086354,0.00008431304],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5198535,0.005048067,0.4404499,0.001997203,0.0005719792,0.0002251644,0.0004276515,0.01602899,0.01539758],"genre_scores_gemma":[0.9415538,0.0006249355,0.05297644,0.0002179772,0.00005359799,0.0001056685,0.0003031571,0.0001711996,0.003993199],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002246613,"threshold_uncertainty_score":0.006670654,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2281494333","doi":"10.14778/2732977.2732980","title":"An experimental comparison of pregel-like graph processing systems","year":2014,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Graph Theory and Algorithms","field":"Computer Science","cited_by":194,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; PageRank; Graph; Global Positioning System; Theoretical computer science; Operating system","authors":[{"name":"Minyang Han","is_ca":true},{"name":"Khuzaima Daudjee","is_ca":true},{"name":"Khaled Ammar","is_ca":true},{"name":"M. TAMER ÖZSU","is_ca":true},{"name":"Xingfang Wang","is_ca":true},{"name":"Tianqi Jin","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01446843168368664,"gpt":0.2619565896032245,"spread":0.2474881579195378,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002942332,0.0009901201,0.0007837994,0.001370539,0.001074618,0.001304857,0.002257702,0.001041106,0.005131413],"category_scores_gemma":[0.01514348,0.0004561732,0.0004329174,0.002255257,0.001165439,0.003750167,0.001443139,0.001396111,0.001598715],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00114925,"about_ca_system_score_gemma":0.001108922,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003723283,"about_ca_topic_score_gemma":0.004334174,"domain_scores_codex":[0.996111,0.000976745,0.0004185251,0.0008782812,0.001162424,0.000453091],"domain_scores_gemma":[0.9811161,0.009237604,0.0006786254,0.004518917,0.00353776,0.0009110452],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01845503,0.01262326,0.02545566,0.005010725,0.001191484,0.0008346339,0.002395611,0.271425,0.140011,0.02430998,0.1219485,0.376339],"study_design_scores_gemma":[0.002178128,0.01621729,0.04262516,0.0001929922,0.0003899345,0.0007900703,0.003043364,0.6974812,0.1731431,0.01885989,0.04481376,0.000265045],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9606584,0.0009395862,0.01328007,0.0006226831,0.0004130526,0.000385699,0.002066679,0.008253878,0.01337993],"genre_scores_gemma":[0.9405051,0.0004830335,0.04441259,0.0003096818,0.0001005641,0.0003324647,0.008400084,0.0008593379,0.004597082],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005131413,"threshold_uncertainty_score":0.01716632,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2798664493","doi":"10.14778/3192965.3192973","title":"Table union search on open data","year":2018,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":193,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Table (database); Computer science; Benchmark (surveying); Data mining; Set (abstract data type); Semantic search; Domain (mathematical analysis); Ontology; Decision table; Probabilistic logic; Information retrieval; Search engine; Artificial intelligence; Mathematics; Programming language","authors":[{"name":"Fatemeh Nargesian","is_ca":true},{"name":"Erkang Zhu","is_ca":true},{"name":"Ken Q. Pu","is_ca":false},{"name":"Renée J. Miller","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4805425545209558,"gpt":0.4884253644300699,"spread":0.007882809909114052,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00439759,0.0009174236,0.002178551,0.005131835,0.001949731,0.004069827,0.003392497,0.001951023,0.005277407],"category_scores_gemma":[0.028504,0.0009309381,0.002537436,0.01051983,0.001648884,0.009212591,0.004871484,0.001543815,0.001107816],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00163729,"about_ca_system_score_gemma":0.002807291,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006355432,"about_ca_topic_score_gemma":0.006847159,"domain_scores_codex":[0.9944501,0.001424897,0.0005030108,0.001389422,0.001759829,0.000472753],"domain_scores_gemma":[0.982085,0.01225082,0.001197078,0.002703825,0.001291914,0.0004713486],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0008764245,0.0005574672,0.01557208,0.001270303,0.0004438231,0.001172553,0.00131813,0.4051214,0.004733785,0.1075455,0.0332072,0.4281812],"study_design_scores_gemma":[0.00009483257,0.0001152409,0.001094164,0.0001117102,0.00009716157,0.0004965825,0.0005219899,0.8407168,0.004185756,0.142105,0.01041955,0.00004125666],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08211512,0.002759091,0.8976465,0.001390047,0.0001406111,0.000336657,0.004987122,0.005803411,0.004821426],"genre_scores_gemma":[0.3087517,0.0007467618,0.6779891,0.0004006174,0.00009948557,0.0003581535,0.008297247,0.0005445703,0.002812491],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006355432,"threshold_uncertainty_score":0.02325696,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2751694342","doi":"10.14778/3137628.3137630","title":"Trajectory similarity join in spatial networks","year":2017,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":179,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Kootenay Association for Science & Technology","funders":"Beijing Nova Program; King Abdullah University of Science and Technology; National Natural Science Foundation of China; Innovationsfonden","keywords":"Join (topology); Computer science; Pruning; Similarity (geometry); Trajectory; Nearest neighbor search; Heuristic; Matching (statistics); Data mining; Scheduling (production processes); Algorithm; Theoretical computer science; Artificial intelligence; Mathematics; Mathematical optimization","authors":[{"name":"Shuo Shang","is_ca":true},{"name":"Lisi Chen","is_ca":false},{"name":"Zhewei Wei","is_ca":false},{"name":"Christian S. Jensen","is_ca":false},{"name":"Kai Zheng","is_ca":false},{"name":"Panos Kalnis","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01795752554889163,"gpt":0.2326463657509419,"spread":0.2146888402020503,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002328805,0.0008499649,0.001586289,0.003559474,0.00172811,0.002500695,0.002268204,0.001384125,0.002884897],"category_scores_gemma":[0.01002742,0.0005220327,0.001092477,0.005867735,0.001128617,0.005080509,0.003660184,0.001085519,0.0009060127],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001841219,"about_ca_system_score_gemma":0.001955711,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008056411,"about_ca_topic_score_gemma":0.006782762,"domain_scores_codex":[0.9965901,0.0006906189,0.0002682194,0.0009711265,0.001224424,0.0002554696],"domain_scores_gemma":[0.9961858,0.001619811,0.0005216415,0.000822759,0.0006227158,0.0002271729],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007634218,0.000309154,0.009325427,0.0004037155,0.0002460965,0.0004799389,0.0007790136,0.5982834,0.007538519,0.09688129,0.007754865,0.2772351],"study_design_scores_gemma":[0.00003932925,0.00009898534,0.001150754,0.00002310965,0.00003766466,0.0002515283,0.0002579213,0.9164852,0.004638363,0.07045581,0.006537742,0.00002361026],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06919332,0.001341047,0.922861,0.0004873965,0.00007873572,0.0002716003,0.001151843,0.001394408,0.003220666],"genre_scores_gemma":[0.5017853,0.0009067567,0.488082,0.0001706108,0.000148851,0.0003003645,0.00370589,0.0001992821,0.004700939],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008056411,"threshold_uncertainty_score":0.01601905,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2005499394","doi":"10.14778/2168651.2168658","title":"Dense subgraph maintenance under streaming edge weight updates for real-time story identification","year":2012,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":172,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Enhanced Data Rates for GSM Evolution; Identification (biology); Social media; Globe; Scale (ratio); Point (geometry); Data science; Edge device; Range (aeronautics); Social network (sociolinguistics); World Wide Web; Artificial intelligence; Geography; Mathematics; Engineering","authors":[{"name":"Albert Angel","is_ca":true},{"name":"Nikos Sarkas","is_ca":true},{"name":"Nick Koudas","is_ca":true},{"name":"Divesh Srivastava","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0132090247020189,"gpt":0.2384151435654986,"spread":0.2252061188634797,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001517123,0.0009236693,0.001450579,0.002850521,0.0007931802,0.001218835,0.003024932,0.001241139,0.00103636],"category_scores_gemma":[0.01736637,0.0007298542,0.0006752348,0.003270765,0.0008244656,0.00416486,0.001646877,0.001088803,0.0005256925],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009957345,"about_ca_system_score_gemma":0.001067874,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006720096,"about_ca_topic_score_gemma":0.01021115,"domain_scores_codex":[0.9988111,0.0002821934,0.0001037186,0.0003674589,0.0003314287,0.0001040963],"domain_scores_gemma":[0.991293,0.004996122,0.001146479,0.001515558,0.0007943123,0.0002544668],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007717399,0.0003907945,0.0152247,0.0003176146,0.0001963684,0.0004400513,0.0006669966,0.3944598,0.01724005,0.008579812,0.008912885,0.5527992],"study_design_scores_gemma":[0.00002887312,0.0000539496,0.001239196,0.000008827959,0.00002933911,0.0001529856,0.00008766979,0.9860271,0.00385686,0.007543304,0.0009612164,0.00001063115],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1518908,0.0008149408,0.8400802,0.000516991,0.00005622228,0.0002299817,0.001092897,0.004168876,0.0011491],"genre_scores_gemma":[0.604866,0.0003570198,0.3890345,0.0001449346,0.0001144436,0.0002642142,0.003451779,0.0003094155,0.001457663],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006720096,"threshold_uncertainty_score":0.01336199,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2133246278","doi":"10.14778/1453856.1453895","title":"Efficient search for the top-k probable nearest neighbors in uncertain databases","year":2008,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":149,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Data mining; Query optimization; Online aggregation; k-nearest neighbors algorithm; Query language; Semantics (computer science); Information retrieval; Sargable; Object (grammar); Point (geometry); Database; Feature (linguistics); Web query classification; Web search query; Search engine; Artificial intelligence","authors":[{"name":"George Beskales","is_ca":true},{"name":"Mohamed A. Soliman","is_ca":true},{"name":"Ihab F. Ilyas","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05346318793657699,"gpt":0.2715280562480217,"spread":0.2180648683114447,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003984306,0.0009378948,0.003364607,0.004446898,0.001646222,0.003350858,0.003866679,0.002307604,0.001635337],"category_scores_gemma":[0.02511799,0.001014346,0.0009179089,0.006381348,0.0009277032,0.00732142,0.002427203,0.001219946,0.00058644],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001354977,"about_ca_system_score_gemma":0.001827491,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007188892,"about_ca_topic_score_gemma":0.01182296,"domain_scores_codex":[0.9953053,0.001268094,0.0006558885,0.001016104,0.001424462,0.0003301212],"domain_scores_gemma":[0.9871206,0.009335232,0.0007983377,0.001407169,0.001049508,0.0002892043],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001739551,0.0006127961,0.01521179,0.0006535663,0.0003374173,0.0005832411,0.001135874,0.4595737,0.006966825,0.02360196,0.01573184,0.4738514],"study_design_scores_gemma":[0.00005500209,0.0000708286,0.0007603156,0.00001827822,0.00004516274,0.0002457401,0.0003434478,0.9706444,0.002115364,0.0247543,0.0009205624,0.00002666008],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1757375,0.003142703,0.815048,0.00121939,0.00006339722,0.0001918292,0.001227856,0.001691636,0.001677764],"genre_scores_gemma":[0.4844283,0.0006150185,0.5111876,0.0001871843,0.0001036567,0.0001381096,0.00223492,0.0001832989,0.0009218142],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007188892,"threshold_uncertainty_score":0.02107126,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2263157912","doi":"10.14778/2777598.2777604","title":"Giraph unchained","year":2015,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Graph Theory and Algorithms","field":"Computer Science","cited_by":145,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Asynchronous communication; Computer science; Scalability; Computation; Synchronization (alternating current); Distributed computing; Bulk synchronous parallel; Graph; Parallel computing; Model of computation; Theoretical computer science; Algorithm; Computer network; Operating system","authors":[{"name":"Minyang Han","is_ca":true},{"name":"Khuzaima Daudjee","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01784287209176177,"gpt":0.2045423580511298,"spread":0.186699485959368,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008898036,0.0009498689,0.0005114599,0.0008898854,0.0009648312,0.001621802,0.002527288,0.0008274679,0.03614961],"category_scores_gemma":[0.00371105,0.0007601601,0.000698999,0.0008320847,0.000949964,0.002846458,0.003092166,0.001449063,0.01751194],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007970097,"about_ca_system_score_gemma":0.00140635,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003738397,"about_ca_topic_score_gemma":0.004241342,"domain_scores_codex":[0.99888,0.0001983027,0.00005203603,0.0003854765,0.0003326828,0.0001515285],"domain_scores_gemma":[0.9975011,0.0003253622,0.00006185065,0.001562279,0.0003954964,0.0001538913],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002802666,0.0004121504,0.004937518,0.001177407,0.0002076484,0.0004404066,0.0008129368,0.03496903,0.04192726,0.1180406,0.4265827,0.3676897],"study_design_scores_gemma":[0.0005006552,0.0003870334,0.002884306,0.000134634,0.0001240639,0.0006624658,0.0002227802,0.2142584,0.06176837,0.1023443,0.6165552,0.0001577307],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"software","genre_gemma":"empirical","genre_scores_codex":[0.04708662,0.0009178909,0.3444376,0.0008246782,0.0008671346,0.0004965434,0.006878052,0.4479538,0.1505377],"genre_scores_gemma":[0.4396507,0.0008581638,0.3707804,0.001507328,0.0002539694,0.0009523212,0.02697431,0.03543712,0.1235856],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03614961,"threshold_uncertainty_score":0.1209325,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3085364681","doi":"10.14778/3415478.3415562","title":"Data collection and quality challenges for deep learning","year":2020,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":145,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Machine learning; Artificial intelligence; Feature engineering; Deep learning; Data collection; Big data; Data science; Software; Data mining","authors":[{"name":"Steven Euijong Whang","is_ca":true},{"name":"Jae-Gil Lee","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1284582489283515,"gpt":0.3110977474383105,"spread":0.182639498509959,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05710194,0.00112651,0.002032391,0.004169252,0.002830537,0.009194133,0.005085642,0.00294739,0.003655462],"category_scores_gemma":[0.2077572,0.00137614,0.00154539,0.007496398,0.00409996,0.009662429,0.008774134,0.008299758,0.002638957],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003892775,"about_ca_system_score_gemma":0.008029495,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005440305,"about_ca_topic_score_gemma":0.006050148,"domain_scores_codex":[0.9378331,0.02391522,0.005539044,0.005116389,0.0264243,0.001171815],"domain_scores_gemma":[0.7931804,0.1000525,0.009855264,0.03990825,0.0531601,0.003843497],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004928268,0.0004301,0.02933097,0.003740502,0.0004032459,0.0004381976,0.002049126,0.02005059,0.007714521,0.08695638,0.1157516,0.7326419],"study_design_scores_gemma":[0.0001610188,0.0004631849,0.02111007,0.003583948,0.000193628,0.001066379,0.003058152,0.102932,0.02128894,0.3995869,0.4462947,0.0002610633],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01913211,0.01141685,0.9078093,0.03980681,0.001741266,0.001254668,0.008394149,0.003763807,0.006681067],"genre_scores_gemma":[0.1118938,0.009625044,0.8412076,0.008286663,0.001994267,0.003711052,0.01633989,0.001847473,0.005094176],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.05710194,"threshold_uncertainty_score":0.3019876,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4255889623","doi":"10.1145/3186728.3164139","title":"The ubiquity of large graphs and surprising challenges of graph processing","year":2017,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Visualization and Analytics","field":"Computer Science","cited_by":145,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Suite; Scalability; Visualization; Graph; Software; Data science; Theoretical computer science; World Wide Web; Data mining; Programming language; Database","authors":[{"name":"Siddhartha Sahu","is_ca":true},{"name":"Amine Mhedhbi","is_ca":true},{"name":"Semih Salihoğlu","is_ca":true},{"name":"Jimmy Lin","is_ca":true},{"name":"M. TAMER ÖZSU","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03165215848905461,"gpt":0.305877498930536,"spread":0.2742253404414814,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01700465,0.000594061,0.0006192609,0.003853207,0.002008102,0.005345271,0.001958362,0.002045353,0.002530016],"category_scores_gemma":[0.09851131,0.0009708655,0.0007336288,0.005026787,0.003872423,0.01460275,0.003248434,0.002793205,0.0008190684],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001399061,"about_ca_system_score_gemma":0.001119767,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001565127,"about_ca_topic_score_gemma":0.003373236,"domain_scores_codex":[0.9816841,0.009347416,0.000726993,0.002397612,0.005373306,0.0004704607],"domain_scores_gemma":[0.8101267,0.1614566,0.006602038,0.01110934,0.008404965,0.002300342],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004139372,0.0001863363,0.07200332,0.007270002,0.0003252301,0.003059673,0.09332798,0.01081467,0.01999756,0.09620959,0.06815118,0.6282406],"study_design_scores_gemma":[0.00005189896,0.0003262352,0.05978908,0.001640628,0.0001238301,0.009185734,0.08512022,0.03606024,0.01021767,0.30083,0.4962796,0.0003747606],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5449855,0.02413753,0.3449883,0.05587446,0.0007666642,0.0003722096,0.002189488,0.00424297,0.02244288],"genre_scores_gemma":[0.7618042,0.01273547,0.2118086,0.004253594,0.0008856765,0.0003018294,0.002268744,0.002382689,0.003559127],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01700465,"threshold_uncertainty_score":0.0899303,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3198515637","doi":"10.14778/3476249.3476283","title":"SlimChain","year":2021,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":143,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Blockchain; Scalability; Distributed computing; Database transaction; Computer network; Robustness (evolution); Distributed data store; Node (physics); Computer security; Database","authors":[{"name":"Cheng Xu","is_ca":true},{"name":"Ce Zhang","is_ca":false},{"name":"Jianliang Xu","is_ca":false},{"name":"Jian Pei","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.007848441708610996,"gpt":0.2075309755193116,"spread":0.1996825338107006,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008915981,0.0005439013,0.0005961919,0.0008762743,0.00133056,0.001693665,0.001452409,0.0008521996,0.03530168],"category_scores_gemma":[0.002525793,0.0003702161,0.0004194959,0.001301867,0.0006580763,0.002764096,0.002683704,0.001150539,0.01153806],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008456084,"about_ca_system_score_gemma":0.00219982,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002614754,"about_ca_topic_score_gemma":0.003431875,"domain_scores_codex":[0.9989451,0.0001769741,0.00007703304,0.0001763482,0.0004115542,0.0002130273],"domain_scores_gemma":[0.9982717,0.0002298071,0.0001196923,0.0006941654,0.0004707007,0.0002139533],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001481191,0.0004206042,0.003923547,0.00137039,0.0001351294,0.001041124,0.0005315191,0.05271278,0.02887072,0.2285662,0.1716298,0.5093172],"study_design_scores_gemma":[0.0004312294,0.0004781466,0.0007939258,0.0001975849,0.00007038291,0.0006505815,0.0001672633,0.2930702,0.02709803,0.1190995,0.5578431,0.0001000151],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05609741,0.002933237,0.6674538,0.001875611,0.001544512,0.001798421,0.005697579,0.03787671,0.2247226],"genre_scores_gemma":[0.6090364,0.002798017,0.2490484,0.001242002,0.0004242036,0.001528248,0.01632185,0.001932631,0.1176682],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03530168,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2963174348","doi":"10.14778/2994509.2994534","title":"LSH ensemble","year":2016,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Advanced Image and Video Retrieval Techniques","field":"Computer Science","cited_by":136,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Jaccard index; Computer science; Data mining; Domain (mathematical analysis); Locality-sensitive hashing; Data structure; Data set; Set (abstract data type); Hash function; Mathematics; Cluster analysis; Hash table; Artificial intelligence","authors":[{"name":"Erkang Zhu","is_ca":true},{"name":"Fatemeh Nargesian","is_ca":true},{"name":"Ken Q. Pu","is_ca":false},{"name":"Renée J. Miller","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01131773862150246,"gpt":0.2360180883857376,"spread":0.2247003497642351,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001940913,0.001017928,0.002086488,0.002299715,0.001130173,0.00142486,0.002814992,0.001649347,0.006213341],"category_scores_gemma":[0.009290032,0.0004235014,0.0009942448,0.00249552,0.0006770555,0.003554343,0.00305159,0.001580052,0.002750023],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001120628,"about_ca_system_score_gemma":0.00150527,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004156455,"about_ca_topic_score_gemma":0.00725128,"domain_scores_codex":[0.9979661,0.0004088411,0.00009223093,0.000509106,0.0008025373,0.0002212636],"domain_scores_gemma":[0.9961525,0.001461089,0.000194196,0.001197182,0.0007620837,0.0002329961],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002908639,0.0003285727,0.009156119,0.0002516903,0.000272901,0.0001637288,0.0002149887,0.3901968,0.004409726,0.02080715,0.02733235,0.5465751],"study_design_scores_gemma":[0.00001607834,0.00009887708,0.0006630596,0.00001396177,0.00002150712,0.0001257133,0.0000672997,0.9754544,0.002018864,0.01592452,0.005579961,0.00001567707],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08015309,0.002521345,0.8994592,0.0008972928,0.0004385928,0.0003125258,0.002067504,0.004608433,0.009542054],"genre_scores_gemma":[0.4733479,0.0008565969,0.5023776,0.0009670212,0.0005542141,0.0004118701,0.007565315,0.0006200852,0.01329939],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006213341,"threshold_uncertainty_score":0.02078569,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2080132606","doi":"10.14778/1938545.1938547","title":"Automatic wrappers for large scale web extraction","year":2011,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":135,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Noise (video); Scale (ratio); Noisy data; Extraction (chemistry); Data extraction; Data mining; Information extraction; Training set; Artificial intelligence; Machine learning; Information retrieval; Pattern recognition (psychology)","authors":[{"name":"Nilesh Dalvi","is_ca":false},{"name":"Ravi Kumar","is_ca":false},{"name":"Mohamed A. Soliman","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02415002081639386,"gpt":0.2448207655012949,"spread":0.2206707446849011,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002602196,0.001668507,0.001613682,0.004270602,0.001297967,0.002185107,0.001884611,0.001476621,0.003021597],"category_scores_gemma":[0.01139046,0.001200313,0.001828758,0.004085403,0.0008858842,0.004344186,0.003520199,0.002041779,0.006799802],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004966936,"about_ca_system_score_gemma":0.001346164,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000926681,"about_ca_topic_score_gemma":0.001707474,"domain_scores_codex":[0.9972093,0.0006067707,0.0004008726,0.0007326978,0.0008799857,0.0001704273],"domain_scores_gemma":[0.992426,0.001960974,0.000627239,0.003818888,0.00102461,0.0001421712],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002291385,0.0003347528,0.005182769,0.000811688,0.0003316313,0.0008256119,0.0005330854,0.02544559,0.04993481,0.01702368,0.04627993,0.8530673],"study_design_scores_gemma":[0.00006642007,0.000129073,0.003284306,0.0002080914,0.0002187438,0.001336194,0.000190671,0.6150351,0.2012489,0.09832586,0.07982334,0.0001333824],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003154902,0.0001929454,0.9661502,0.00007452469,0.00003154033,0.00009006762,0.0007244122,0.02909285,0.000488524],"genre_scores_gemma":[0.04740198,0.0002877382,0.9415964,0.0001583595,0.00006681975,0.0001996986,0.005358897,0.003137903,0.00179219],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004270602,"threshold_uncertainty_score":0.01376188,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2110020044","doi":"10.14778/1453856.1453922","title":"A practical scalable distributed B-tree","year":2008,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Distributed systems and fault tolerance","field":"Computer Science","cited_by":135,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Distributed computing; Scalability; Tree (set theory); Fault tolerance; Distributed transaction; Concurrency; Transaction processing; Database transaction; Operating system; Database","authors":[{"name":"Marcos K. Aguilera","is_ca":false},{"name":"Wojciech Golab","is_ca":true},{"name":"Mehul A. Shah","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02628965727836388,"gpt":0.2516244494746975,"spread":0.2253347921963336,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001158021,0.0003841064,0.0005687174,0.0006775389,0.001266788,0.001317321,0.002266335,0.001265629,0.01020817],"category_scores_gemma":[0.003908471,0.0003989204,0.000417063,0.001630831,0.0005215264,0.002423637,0.002203713,0.00103567,0.004032465],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006705831,"about_ca_system_score_gemma":0.001676923,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002046647,"about_ca_topic_score_gemma":0.002254084,"domain_scores_codex":[0.9987914,0.0001578712,0.00009432388,0.0001948594,0.0006473972,0.0001142558],"domain_scores_gemma":[0.9985032,0.0002855576,0.00006812212,0.0003755016,0.0005924118,0.0001752325],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004175011,0.0002523889,0.001450146,0.0006531661,0.000060803,0.0006378749,0.0002455833,0.1139954,0.04514711,0.1540456,0.09570391,0.5873906],"study_design_scores_gemma":[0.0002989906,0.0003298957,0.0004128127,0.00007969943,0.00004566594,0.0009744731,0.0001155121,0.6860753,0.01255356,0.1089796,0.1900767,0.00005788406],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00675231,0.0005427831,0.9780831,0.0008433866,0.0001353193,0.0002084628,0.0002916799,0.00399447,0.009148495],"genre_scores_gemma":[0.08324079,0.0004819768,0.9094471,0.0003172563,0.00009595822,0.0002931757,0.0007589291,0.0002483085,0.005116498],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01020817,"threshold_uncertainty_score":0.03414971,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3012550338","doi":"10.14778/3389133.3389134","title":"Dash","year":2020,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Advanced Data Storage Technologies","field":"Computer Science","cited_by":135,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Hash function; Scalability; Emulation; Hash table; Factor (programming language); Double hashing; Table (database)","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.02138445553614614,"gpt":0.2242101176233977,"spread":0.2028256620872516,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009530944,0.0008403062,0.0006575534,0.0006708308,0.0007925875,0.002582564,0.002693187,0.0008178233,0.06572507],"category_scores_gemma":[0.002924285,0.0005792958,0.0004244682,0.000874974,0.0006065201,0.004031744,0.00365488,0.001382399,0.04104789],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006738437,"about_ca_system_score_gemma":0.001199097,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009339538,"about_ca_topic_score_gemma":0.001414902,"domain_scores_codex":[0.9988183,0.0001170549,0.0001237667,0.0002260289,0.0005601695,0.0001545752],"domain_scores_gemma":[0.9976971,0.0002493751,0.00008245679,0.00098754,0.000789742,0.0001938416],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002757903,0.0004765422,0.00647753,0.001387156,0.0001646946,0.00048744,0.0005473629,0.005680885,0.07638193,0.07103008,0.4231462,0.4114622],"study_design_scores_gemma":[0.0001872192,0.0004052103,0.001279336,0.00007340479,0.00005320411,0.0005147141,0.0002538211,0.02867669,0.08011591,0.01759726,0.8707415,0.0001017528],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"methods","genre_scores_codex":[0.08247387,0.00373077,0.3182985,0.0025802,0.003840753,0.001179286,0.03173808,0.2366743,0.3194843],"genre_scores_gemma":[0.3917081,0.002877878,0.1573106,0.002218698,0.0005380669,0.001173914,0.073356,0.0122109,0.3586058],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.06572507,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2619906413","doi":"10.14778/3099622.3099623","title":"Revisiting the stop-and-stare algorithms for influence maximization","year":2017,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Complex Network Analysis Techniques","field":"Physics and Astronomy","cited_by":132,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Ministry of Education, India","keywords":"Maximization; Scalability; Computer science; Set (abstract data type); Scaling; Order (exchange); Approximation algorithm; Mathematical optimization; Algorithm; Mathematics; Economics","authors":[{"name":"Keke Huang","is_ca":false},{"name":"Sibo Wang","is_ca":false},{"name":"Glenn S. Bevilacqua","is_ca":true},{"name":"Xiaokui Xiao","is_ca":false},{"name":"Laks V. S. Lakshmanan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0194310404711445,"gpt":0.2842343860065044,"spread":0.2648033455353598,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006935547,0.001851465,0.0024056,0.002099448,0.001507636,0.003103946,0.002737874,0.002208945,0.005249097],"category_scores_gemma":[0.03871835,0.0008947405,0.002217995,0.002561452,0.002486167,0.006410138,0.003535455,0.004086678,0.002220964],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001942327,"about_ca_system_score_gemma":0.0030999,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004252749,"about_ca_topic_score_gemma":0.00765274,"domain_scores_codex":[0.9953805,0.002116367,0.0002348956,0.0008689222,0.001044184,0.0003551218],"domain_scores_gemma":[0.9713637,0.02077942,0.0008785013,0.004399718,0.001924522,0.0006540426],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004800928,0.0003362755,0.004663488,0.0005103163,0.0002441548,0.000170912,0.0006191935,0.4953064,0.004063027,0.1714075,0.01998568,0.3022128],"study_design_scores_gemma":[0.00003517349,0.00007071879,0.0002226632,0.00003281846,0.00002397126,0.0000773233,0.00004656501,0.9153221,0.001110506,0.07859246,0.004452045,0.0000136351],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0112829,0.0008377076,0.9790261,0.001212959,0.0001472214,0.0001227756,0.0002169809,0.001057008,0.006096295],"genre_scores_gemma":[0.3302193,0.001187344,0.659142,0.001111098,0.0006295267,0.0003410921,0.0008054327,0.000831301,0.005732923],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006935547,"threshold_uncertainty_score":0.03667915,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2162783807","doi":"10.14778/2021017.2021025","title":"Keyword search in graphs","year":2011,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":132,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"","keywords":"Substructure; Clique; Computer science; Graph; Combinatorics; Theoretical computer science; Time complexity; Mathematics; Algorithm","authors":[{"name":"Mehdi Kargar","is_ca":true},{"name":"Aijun An","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03651081320407087,"gpt":0.2195111061961708,"spread":0.1830002929920999,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00123572,0.0009957463,0.001601656,0.00313985,0.0015226,0.003679835,0.001988003,0.002024083,0.006952313],"category_scores_gemma":[0.01115042,0.0009229629,0.001308278,0.007795485,0.001317765,0.008909615,0.002432011,0.001296253,0.002922565],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002036925,"about_ca_system_score_gemma":0.001629392,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004747696,"about_ca_topic_score_gemma":0.004953087,"domain_scores_codex":[0.9972768,0.000910157,0.0002404823,0.0008716729,0.0004500738,0.0002507854],"domain_scores_gemma":[0.9939927,0.004116666,0.0005052529,0.0007695441,0.0004229021,0.0001928289],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006277694,0.0002809465,0.003163277,0.002888471,0.0002971029,0.0008572183,0.00122683,0.2196971,0.01190472,0.3166703,0.05502875,0.3873575],"study_design_scores_gemma":[0.00009206554,0.0001053622,0.0006494442,0.0001269249,0.00007200029,0.0009228826,0.000433453,0.2843508,0.004845701,0.6707673,0.03758225,0.00005171974],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04531558,0.005098478,0.9254759,0.002606931,0.00017137,0.0004927976,0.00402356,0.002968161,0.0138472],"genre_scores_gemma":[0.2860639,0.004916594,0.6914189,0.0008215114,0.0002415799,0.0003756056,0.006168046,0.0005171457,0.009476676],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006952313,"threshold_uncertainty_score":0.02325779,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2160152607","doi":"10.14778/1687627.1687754","title":"Efficient method for maximizing bichromatic reverse nearest neighbor","year":2009,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":129,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; k-nearest neighbors algorithm; Point (geometry); Best bin first; Exponential function; Algorithm; Exponential growth; Theoretical computer science; Mathematics; Artificial intelligence","authors":[{"name":"Raymond Chi-Wing Wong","is_ca":false},{"name":"M. TAMER ÖZSU","is_ca":true},{"name":"Philip S. Yu","is_ca":false},{"name":"Ada Wai-Chee Fu","is_ca":false},{"name":"Lian Liu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01458119393423405,"gpt":0.2572641271071092,"spread":0.2426829331728751,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00157206,0.0011313,0.001877277,0.002361787,0.0008605612,0.001022602,0.002674639,0.001167936,0.004400157],"category_scores_gemma":[0.006288352,0.0006356133,0.0008446203,0.002812432,0.0006000947,0.002264846,0.002824784,0.000913428,0.001803275],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009242684,"about_ca_system_score_gemma":0.001659337,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003892747,"about_ca_topic_score_gemma":0.006198694,"domain_scores_codex":[0.997642,0.0005515056,0.0001347962,0.0004520314,0.001029017,0.0001906925],"domain_scores_gemma":[0.9980293,0.0007052377,0.0001990311,0.0003853452,0.0006047926,0.00007631326],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004481573,0.0002318686,0.001836264,0.0003755607,0.00008333039,0.0001567217,0.0002793113,0.1528829,0.01591794,0.01922286,0.009857926,0.7987072],"study_design_scores_gemma":[0.00006724329,0.00009956697,0.0005798119,0.00002553458,0.00002845482,0.0003891127,0.000107585,0.9690306,0.01016738,0.01329293,0.006174479,0.0000373106],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01342903,0.0005591736,0.982529,0.0001304045,0.00004309442,0.000119588,0.0001318259,0.001028171,0.00202968],"genre_scores_gemma":[0.09750411,0.0002296065,0.8994473,0.00008888402,0.00003843769,0.0002012013,0.000406396,0.0001908526,0.001893243],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004400157,"threshold_uncertainty_score":0.01472002,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2112840274","doi":"10.14778/1920841.1920870","title":"Sampling the repairs of functional dependency violations under hard constraints","year":2010,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":127,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Dependency (UML); Class (philosophy); Functional dependency; Context (archaeology); Sampling (signal processing); Relation (database); Data integrity; Variety (cybernetics); Space (punctuation); Data mining; Metric (unit); Theoretical computer science; Relational database; Database; Artificial intelligence","authors":[{"name":"George Beskales","is_ca":true},{"name":"Ihab F. Ilyas","is_ca":true},{"name":"Lukasz Golab","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1812498346188758,"gpt":0.3696657815601582,"spread":0.1884159469412824,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006641858,0.0007913992,0.00146591,0.001428963,0.0007763588,0.001059771,0.00187476,0.00159415,0.001044564],"category_scores_gemma":[0.04814147,0.0006231864,0.001155796,0.001318159,0.001356201,0.002132396,0.001753863,0.001797797,0.0002708242],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000941277,"about_ca_system_score_gemma":0.001745098,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002166295,"about_ca_topic_score_gemma":0.003335025,"domain_scores_codex":[0.9940351,0.002546913,0.0005231496,0.001137419,0.001386441,0.0003710587],"domain_scores_gemma":[0.9459171,0.04035443,0.002856625,0.007591051,0.00261215,0.0006686944],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001354502,0.0005490964,0.02676857,0.000587062,0.0002555318,0.0006628524,0.001289465,0.6412677,0.0165975,0.01659114,0.006223891,0.2878526],"study_design_scores_gemma":[0.00008066854,0.0002045106,0.001781795,0.0000381374,0.00006413247,0.0003125748,0.0003469138,0.9594907,0.01157619,0.02432048,0.001757698,0.00002626009],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3402188,0.0006179192,0.6543396,0.0009551736,0.00005330814,0.0003085658,0.0006021249,0.001720676,0.001183741],"genre_scores_gemma":[0.7143211,0.000178144,0.2826576,0.000211905,0.00004555859,0.000273417,0.001250731,0.0002609746,0.000800503],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006641858,"threshold_uncertainty_score":0.03512597,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2112485457","doi":"10.14778/1687553.1687567","title":"SQL/MapReduce","year":2009,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":127,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"ASTER","funders":"","keywords":"Computer science; SQL; User-defined function; Database; Schema (genetic algorithms); Scalability; NoSQL; Programming language; Query by Example; Information retrieval","authors":[{"name":"Eric Friedman","is_ca":true},{"name":"Peter M. Pawlowski","is_ca":true},{"name":"John Cieslewicz","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.007158007447369725,"gpt":0.213177163393006,"spread":0.2060191559456363,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003637732,0.002319737,0.001225597,0.00150267,0.00101271,0.003688867,0.005880516,0.0009482381,0.0235506],"category_scores_gemma":[0.005844583,0.001269024,0.002185457,0.001664628,0.000776081,0.00333948,0.004268332,0.003546485,0.03204474],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001075222,"about_ca_system_score_gemma":0.003485363,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004295527,"about_ca_topic_score_gemma":0.003250603,"domain_scores_codex":[0.9954478,0.0005650685,0.0005322359,0.000902441,0.002167973,0.0003844705],"domain_scores_gemma":[0.9966191,0.0005944523,0.0001614337,0.001302611,0.00098631,0.0003359623],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000781192,0.0004670609,0.001748765,0.001432041,0.000302819,0.000460885,0.0006348636,0.0112695,0.01758202,0.06268515,0.6631722,0.2394635],"study_design_scores_gemma":[0.000242501,0.0001675073,0.001206847,0.0001012055,0.00005396668,0.0006207684,0.0002781229,0.07568464,0.03206975,0.04948739,0.8398957,0.0001916039],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003648268,0.0005181087,0.5967606,0.0009794378,0.0005347223,0.00110458,0.02295094,0.3411616,0.03234172],"genre_scores_gemma":[0.0767946,0.001723938,0.6958285,0.00289414,0.0005114646,0.002306438,0.1207009,0.05727695,0.04196307],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0235506,"threshold_uncertainty_score":0.07878458,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2970828623","doi":"10.14778/3342263.3342645","title":"Efficient algorithms for densest subgraph discovery","year":2019,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":115,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Induced subgraph isomorphism problem; Intuition; Computer science; Subgraph isomorphism problem; Graph; Algorithm; Color-coding; Efficient algorithm; Theoretical computer science; Artificial intelligence; Line graph","authors":[{"name":"Yixiang Fang","is_ca":false},{"name":"Kaiqiang Yu","is_ca":false},{"name":"Reynold Cheng","is_ca":false},{"name":"Laks V. S. Lakshmanan","is_ca":true},{"name":"Xuemin Lin","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01307049411882908,"gpt":0.2375503892464292,"spread":0.2244798951276001,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002762877,0.002450887,0.002683559,0.005454037,0.001825824,0.002922295,0.004275587,0.002704507,0.006323712],"category_scores_gemma":[0.015302,0.001469997,0.002781273,0.007930484,0.001277073,0.006116601,0.004554974,0.002406863,0.003039423],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002329322,"about_ca_system_score_gemma":0.004490763,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006597212,"about_ca_topic_score_gemma":0.01349676,"domain_scores_codex":[0.9961489,0.0007727728,0.0002957236,0.001166436,0.001172072,0.0004441442],"domain_scores_gemma":[0.990545,0.004801808,0.0007362175,0.002406641,0.001161817,0.000348474],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003835076,0.0006338007,0.004093499,0.0009383431,0.0002894562,0.0002669595,0.0005675493,0.2048263,0.006386471,0.04434923,0.03735498,0.6999099],"study_design_scores_gemma":[0.0001494718,0.00008014928,0.0006908114,0.00004501609,0.00007703641,0.0003581885,0.0001911556,0.870224,0.002918431,0.1166183,0.008617361,0.00003011276],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01313038,0.0009566081,0.9772779,0.0006265506,0.00007726126,0.0003665573,0.001011948,0.003949999,0.002602762],"genre_scores_gemma":[0.07665639,0.0005479236,0.914748,0.0002441538,0.00008888763,0.000437499,0.004659056,0.0004448317,0.002173252],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006597212,"threshold_uncertainty_score":0.02115494,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2571118757","doi":"10.14778/3015274.3015276","title":"Mostly-optimistic concurrency control for highly contended dynamic workloads on a thousand cores","year":2016,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Distributed systems and fault tolerance","field":"Computer Science","cited_by":113,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Concurrency control; Server; Parallel computing; Concurrency; Cache; Distributed computing; Cache coherence; Lock (firearm); Deadlock; Out-of-order execution; Serialization; Operating system; CPU cache; Database transaction; Cache algorithms; Database","authors":[{"name":"Tianzheng Wang","is_ca":true},{"name":"Hideaki Kimura","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01018218956139474,"gpt":0.2339998161680842,"spread":0.2238176266066894,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001577641,0.000656544,0.0004759355,0.0004805995,0.0008446991,0.001114919,0.001577577,0.0002828559,0.001031166],"category_scores_gemma":[0.004224886,0.0003473331,0.0002272773,0.0005930139,0.000811714,0.001372939,0.001015438,0.0007531723,0.0002181784],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007770901,"about_ca_system_score_gemma":0.002550749,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004818294,"about_ca_topic_score_gemma":0.006544415,"domain_scores_codex":[0.998418,0.0002554427,0.0001228404,0.0002149447,0.0007040303,0.0002848579],"domain_scores_gemma":[0.9966369,0.0009183764,0.0004136282,0.001139216,0.0006372695,0.0002546463],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001933364,0.0005367618,0.02545721,0.0003704719,0.000158711,0.0005482102,0.001061992,0.2789944,0.2780093,0.02183542,0.007989328,0.3831048],"study_design_scores_gemma":[0.0001175737,0.0003794762,0.002255775,0.00002523166,0.00004660269,0.0002164148,0.0001661189,0.9101683,0.07673819,0.00549571,0.004342298,0.00004825138],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6771111,0.001897215,0.3054152,0.0004502787,0.0001396648,0.0001475149,0.00007773902,0.00788448,0.006876871],"genre_scores_gemma":[0.9693125,0.0001559821,0.02914934,0.00007615101,0.00002284677,0.00003829215,0.00006245211,0.00008965648,0.001092745],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004818294,"threshold_uncertainty_score":0.009580493,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2154768509","doi":"10.14778/1453856.1453934","title":"Efficient network aware search in collaborative tagging sites","year":2008,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":112,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Popularity; Cluster analysis; Seekers; Context (archaeology); Upper and lower bounds; Heuristic; Space (punctuation); Information retrieval; Data mining; Machine learning; Artificial intelligence; Mathematics; Geography","authors":[{"name":"Sihem Amer Yahia","is_ca":false},{"name":"Michael Benedikt","is_ca":false},{"name":"Laks V. S. Lakshmanan","is_ca":true},{"name":"Julia Stoyanovich","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01881665465521546,"gpt":0.234770653360951,"spread":0.2159539987057356,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003343349,0.0008142266,0.002247597,0.003186424,0.001752146,0.002921827,0.002772148,0.001806523,0.001515398],"category_scores_gemma":[0.01742416,0.0007578142,0.000850577,0.005506529,0.00119131,0.005478757,0.002698744,0.000906485,0.0009563499],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001555894,"about_ca_system_score_gemma":0.001614328,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005331591,"about_ca_topic_score_gemma":0.009775597,"domain_scores_codex":[0.9971719,0.0009333146,0.0002133403,0.0006724093,0.0006092726,0.0003997342],"domain_scores_gemma":[0.989407,0.00634803,0.001103732,0.001869678,0.0008283364,0.0004432292],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001702193,0.0006145177,0.02922441,0.0007888292,0.0003021344,0.0005879742,0.002218715,0.587754,0.02310791,0.04567719,0.01101087,0.2970113],"study_design_scores_gemma":[0.00005298417,0.0001070943,0.001653003,0.00001774126,0.00005585298,0.0002309767,0.0003338365,0.9607445,0.004027523,0.03103467,0.001715783,0.00002599858],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3467378,0.001554494,0.6432123,0.0007131455,0.00004540091,0.0002427851,0.0008934593,0.001873292,0.004727425],"genre_scores_gemma":[0.7774791,0.0003591123,0.2172243,0.0001038446,0.00006549033,0.0001233702,0.00128899,0.0001815646,0.003174167],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005331591,"threshold_uncertainty_score":0.01768148,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2106019582","doi":"10.14778/2168651.2168659","title":"ReStore","year":2012,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Cloud Computing and Resource Management","field":"Computer Science","cited_by":110,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Workflow; Dataflow; Reuse; Compiler; Implementation; Distributed computing; Operating system; Database; Parallel computing; Programming language","authors":[{"name":"Iman Elghandour","is_ca":true},{"name":"Ashraf Aboulnaga","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01353766561554491,"gpt":0.214237238605279,"spread":0.2006995729897341,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00148278,0.001445578,0.0008916596,0.001964979,0.001509628,0.003669162,0.003339722,0.001229073,0.08669438],"category_scores_gemma":[0.00577869,0.0008828519,0.001355137,0.001601181,0.001053896,0.004963222,0.005616363,0.002224003,0.07582542],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001018965,"about_ca_system_score_gemma":0.002293861,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003238323,"about_ca_topic_score_gemma":0.003511785,"domain_scores_codex":[0.997925,0.0001673628,0.0001206711,0.0005408972,0.0009476177,0.0002984022],"domain_scores_gemma":[0.9965991,0.0003574345,0.0001404981,0.001828463,0.0008411118,0.000233266],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007449122,0.0002322114,0.003132307,0.0007524079,0.00009696987,0.0002999851,0.0005116641,0.004112348,0.01021166,0.02380797,0.6692942,0.2868033],"study_design_scores_gemma":[0.00008543431,0.00008882695,0.001736652,0.00007953403,0.00004052315,0.0003961402,0.0002048165,0.009924855,0.01706356,0.01412049,0.9561861,0.00007306657],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"software","genre_gemma":"empirical","genre_scores_codex":[0.01796295,0.001361062,0.1890476,0.002470468,0.002142267,0.001029969,0.03296043,0.5624081,0.1906172],"genre_scores_gemma":[0.1718395,0.002140825,0.260179,0.005616973,0.0009056763,0.001299614,0.1549493,0.1018078,0.3012613],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.08669438,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2156972533","doi":"10.14778/1920841.1920874","title":"SECRET","year":2010,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":109,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Semantics (computer science); Key (lock); Variation (astronomy); Range (aeronautics); Window (computing); World Wide Web; Programming language; Computer security; Engineering","authors":[{"name":"Irina Botan","is_ca":false},{"name":"Roozbeh Derakhshan","is_ca":false},{"name":"Nihal Dindar","is_ca":false},{"name":"Laura M. Haas","is_ca":false},{"name":"Renée J. Miller","is_ca":true},{"name":"Nesime Tatbul","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.004724064018066928,"gpt":0.2012143182637794,"spread":0.1964902542457125,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001777709,0.0006213028,0.0005539363,0.001111132,0.0008251129,0.004968903,0.00215876,0.001377278,0.07540943],"category_scores_gemma":[0.007320987,0.0005319561,0.0006505111,0.001218581,0.0008053074,0.00773795,0.002654738,0.001104051,0.04799901],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001227042,"about_ca_system_score_gemma":0.002063441,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002519673,"about_ca_topic_score_gemma":0.001572768,"domain_scores_codex":[0.9982144,0.0002342873,0.0001475825,0.0003761316,0.0008224593,0.0002052403],"domain_scores_gemma":[0.9962876,0.0005956435,0.0002718569,0.001671352,0.0009256158,0.0002478557],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007677595,0.0001239117,0.004996776,0.0005158933,0.00004944202,0.0004149239,0.0005794838,0.008812725,0.004857874,0.495949,0.2909338,0.1919985],"study_design_scores_gemma":[0.00004739066,0.00005439598,0.0005317967,0.0000848311,0.00001819235,0.0004723139,0.00009469346,0.01649855,0.003647977,0.05641769,0.9220969,0.00003536921],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.02134201,0.001768008,0.3278565,0.007424854,0.001745453,0.001013293,0.03285099,0.04686666,0.5591322],"genre_scores_gemma":[0.304287,0.004110622,0.1373003,0.005588649,0.0009455418,0.001156757,0.06285513,0.01130015,0.4724559],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.07540943,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2970408474","doi":"10.14778/3342263.3342274","title":"PrivateSQL","year":2019,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":108,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Differential privacy; SQL; Relational database; Schema (genetic algorithms); Conjunctive query; Information retrieval; Workload; View; Relation (database); Database; Data mining; Database design","authors":[{"name":"Ios Kotsogiannis","is_ca":false},{"name":"Yuchao Tao","is_ca":false},{"name":"Xi He","is_ca":true},{"name":"Maryam Fanaeepour","is_ca":false},{"name":"Ashwin Machanavajjhala","is_ca":false},{"name":"Michael Hay","is_ca":false},{"name":"Gerome Miklau","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01425645443807064,"gpt":0.2277791953859026,"spread":0.213522740947832,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007529411,0.001685657,0.001520714,0.001863372,0.001218554,0.007394466,0.006869661,0.002912393,0.05990458],"category_scores_gemma":[0.02648951,0.001942877,0.002013639,0.002674615,0.002108108,0.0115647,0.01080936,0.004216841,0.04957975],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002482085,"about_ca_system_score_gemma":0.004105178,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005241054,"about_ca_topic_score_gemma":0.00378339,"domain_scores_codex":[0.9918014,0.001777773,0.0009584874,0.001279991,0.003433114,0.0007491318],"domain_scores_gemma":[0.9870236,0.003224904,0.0006799468,0.006698947,0.001958778,0.0004138239],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001652447,0.000207959,0.002901311,0.002102051,0.0002146552,0.0003875324,0.0007881068,0.006795975,0.007260286,0.1704181,0.5715935,0.2356781],"study_design_scores_gemma":[0.0003265774,0.0001343586,0.0007014884,0.0002410677,0.00004846329,0.0004811328,0.0001605943,0.02623712,0.01100864,0.1339788,0.8265503,0.00013154],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.004051033,0.002041333,0.5328277,0.002875245,0.0006489765,0.0008711866,0.05630909,0.3654568,0.03491852],"genre_scores_gemma":[0.1539546,0.004718132,0.3727598,0.0115557,0.0008695481,0.002626244,0.2700798,0.1015282,0.08190805],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.05990458,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2295468252","doi":"10.14778/2856318.2856325","title":"Combining quantitative and logical data cleaning","year":2015,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":104,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Ontario Tech University; McMaster University; University of Toronto","funders":"","keywords":"Computer science; Metric (unit); Inference; Functional dependency; Set (abstract data type); Distortion (music); Statistical inference; Data mining; Algorithm; Theoretical computer science; Dependency (UML); Quality (philosophy); Data quality; Artificial intelligence; Relational database; Mathematics","authors":[{"name":"Nataliya Prokoshyna","is_ca":true},{"name":"Jaroslaw Szlichta","is_ca":true},{"name":"Fei Chiang","is_ca":true},{"name":"Renée J. Miller","is_ca":true},{"name":"Divesh Srivastava","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.6149975953066957,"gpt":0.4683191180089634,"spread":0.1466784772977324,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01963293,0.001763221,0.002002778,0.005336357,0.001624887,0.005092177,0.006389302,0.002015429,0.002566699],"category_scores_gemma":[0.05328358,0.001416168,0.003621856,0.005225926,0.004657819,0.009263154,0.01152399,0.004628743,0.0009114493],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002409616,"about_ca_system_score_gemma":0.004903195,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003013635,"about_ca_topic_score_gemma":0.004353407,"domain_scores_codex":[0.9740335,0.007640894,0.002531967,0.003468806,0.01144813,0.0008767135],"domain_scores_gemma":[0.9354174,0.03012721,0.003291831,0.0236405,0.006916509,0.0006065988],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004300622,0.0004835255,0.01062826,0.001454712,0.0005501051,0.0007005003,0.001254775,0.1696998,0.02619185,0.1658238,0.01186396,0.6109186],"study_design_scores_gemma":[0.00008481284,0.000249364,0.002097468,0.0002416865,0.0002177236,0.0009744816,0.0008237375,0.6208883,0.04097774,0.3036481,0.02963696,0.0001596626],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003592829,0.0001494208,0.9926888,0.0006299151,0.0000323675,0.0001054785,0.0001743955,0.00170846,0.0009182467],"genre_scores_gemma":[0.08399864,0.0001624015,0.9131966,0.0004791341,0.00004791344,0.0001651064,0.0007536365,0.0004406451,0.0007557734],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01963293,"threshold_uncertainty_score":0.1038301,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1656389077","doi":"10.14778/2735479.2735485","title":"Rapid sampling for visualizations with ordering guarantees","year":2015,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Data Visualization and Analytics","field":"Computer Science","cited_by":102,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"National Institute of General Medical Sciences","keywords":"Computer science; Bar chart; Visualization; Focus (optics); Sampling (signal processing); Chart; Theoretical computer science; Property (philosophy); Algorithm; Data mining; Mathematics; Statistics; Computer vision","authors":[{"name":"Albert Kim","is_ca":false},{"name":"Eric Blais","is_ca":true},{"name":"Aditya Parameswaran","is_ca":false},{"name":"Piotr Indyk","is_ca":false},{"name":"Samuel Madden","is_ca":false},{"name":"Ronitt Rubinfeld","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07172555247471732,"gpt":0.3140839767866019,"spread":0.2423584243118846,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006365862,0.001326168,0.00129078,0.001886658,0.001144925,0.003497286,0.001711333,0.001676817,0.007357129],"category_scores_gemma":[0.06189564,0.001072562,0.00112265,0.001940134,0.001684753,0.004954061,0.003515337,0.002581768,0.002221414],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001347405,"about_ca_system_score_gemma":0.001490287,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001875183,"about_ca_topic_score_gemma":0.002427313,"domain_scores_codex":[0.9939621,0.002567356,0.0004137935,0.0008388458,0.00192268,0.0002951667],"domain_scores_gemma":[0.9695964,0.01831299,0.001699146,0.006646705,0.003047841,0.000696874],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001378816,0.0001971074,0.006275727,0.0007865188,0.000124702,0.0004260461,0.001669753,0.2880025,0.03286411,0.2066562,0.01813436,0.4434842],"study_design_scores_gemma":[0.0001286185,0.0001263769,0.0005944074,0.00006409738,0.00001838594,0.0002134642,0.0001958249,0.8038648,0.009892378,0.1742041,0.01066349,0.0000339447],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008890252,0.0002754317,0.9871329,0.0003136294,0.00003894258,0.00009527728,0.0001739614,0.002136116,0.0009434916],"genre_scores_gemma":[0.2197118,0.0004123439,0.7756688,0.0002445635,0.00009830163,0.0004311918,0.0009546572,0.0009663604,0.001512111],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007357129,"threshold_uncertainty_score":0.03366631,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2119323564","doi":"10.14778/2536206.2536208","title":"A data-adaptive and dynamic segmentation index for whole matching on time series","year":2013,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Time Series Analysis and Forecasting","field":"Computer Science","cited_by":99,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Search engine indexing; Series (stratigraphy); Segmentation; Computer science; Matching (statistics); Index (typography); Time series; Similarity (geometry); Tree (set theory); Nearest neighbor search; Data mining; Algorithm; Pattern recognition (psychology); Mathematics; Artificial intelligence; Machine learning; Statistics; Image (mathematics)","authors":[{"name":"Yang Wang","is_ca":false},{"name":"Peng Wang","is_ca":false},{"name":"Jian Pei","is_ca":true},{"name":"Wei Wang","is_ca":false},{"name":"Sheng Huang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0158004004847677,"gpt":0.2299606015603585,"spread":0.2141602010755908,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001957362,0.0006020599,0.001426683,0.004958992,0.0008965181,0.001845068,0.001673453,0.0008962239,0.001979674],"category_scores_gemma":[0.01327511,0.0003535975,0.0006598708,0.008064051,0.0007904985,0.006622171,0.002334304,0.001156188,0.001242524],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001208241,"about_ca_system_score_gemma":0.001811872,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002305513,"about_ca_topic_score_gemma":0.002829862,"domain_scores_codex":[0.9980932,0.0002488084,0.0003088071,0.0004701249,0.0007824068,0.00009670242],"domain_scores_gemma":[0.9959977,0.001302144,0.0004104748,0.001138153,0.0009362596,0.0002153791],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004787286,0.000228838,0.007331142,0.000380214,0.0001105022,0.0001806727,0.0005022344,0.07269931,0.03145682,0.07492201,0.01958707,0.7921225],"study_design_scores_gemma":[0.00005436903,0.0003454705,0.003187095,0.0000594227,0.00007048879,0.000675909,0.0002248961,0.8702545,0.01595977,0.07671402,0.03236268,0.00009134479],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02600142,0.001213434,0.9669055,0.0002289495,0.000141272,0.0001355289,0.001349936,0.001766315,0.002257567],"genre_scores_gemma":[0.211367,0.001086753,0.7805803,0.0001727184,0.0002112085,0.0002905923,0.004262335,0.0002797053,0.001749454],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004958992,"threshold_uncertainty_score":0.0103516,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3111141572","doi":"10.14778/3461535.3461552","title":"Are we ready for learned cardinality estimation?","year":2021,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":94,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Cardinality (data modeling); Inference; Focus (optics); Ask price; Domain (mathematical analysis); Workload; Data modeling","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.07254020723674541,"gpt":0.3182425928154698,"spread":0.2457023855787244,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01029612,0.001413086,0.002231424,0.0009575972,0.0006713409,0.004049168,0.004489764,0.002181523,0.001968587],"category_scores_gemma":[0.07084084,0.001206333,0.001152926,0.001613366,0.002030315,0.01707021,0.00337298,0.005253138,0.001241533],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002344457,"about_ca_system_score_gemma":0.003129833,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005375326,"about_ca_topic_score_gemma":0.006817847,"domain_scores_codex":[0.9907519,0.004175003,0.0005388198,0.001923136,0.002029183,0.0005819459],"domain_scores_gemma":[0.9541237,0.0244292,0.003044044,0.01339665,0.003982592,0.001023758],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006511849,0.0004430125,0.02570666,0.0006423184,0.0003528586,0.0001724881,0.0006470481,0.5115672,0.005639557,0.03961217,0.01997532,0.3945903],"study_design_scores_gemma":[0.00003348662,0.0000715236,0.001021306,0.00005750674,0.00003194176,0.00008474244,0.0001983291,0.9613079,0.002418015,0.03167138,0.003073532,0.00003040959],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0552291,0.002267407,0.9302608,0.005571324,0.0002221499,0.0001336624,0.0006606451,0.003718174,0.001936845],"genre_scores_gemma":[0.5156555,0.001203534,0.4772713,0.001553916,0.000322248,0.000196449,0.001751241,0.0007182777,0.001327641],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01029612,"threshold_uncertainty_score":0.05445176,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3000259868","doi":"10.14778/3372716.3372728","title":"Evaluating persistent memory range indexes","year":2019,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Parallel Computing and Optimization Techniques","field":"Computer Science","cited_by":94,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Benchmarking; Index (typography); Dram; Range (aeronautics); Tree (set theory); Key (lock); Data structure; Focus (optics); Persistent data structure; Database; Operating system; Computer hardware; Programming language","authors":[{"name":"Lucas Lersch","is_ca":false},{"name":"Xiangpeng Hao","is_ca":true},{"name":"Ismail Oukid","is_ca":false},{"name":"Tianzheng Wang","is_ca":true},{"name":"Thomas Willhalm","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03028429458173998,"gpt":0.2794155407627558,"spread":0.2491312461810158,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002810555,0.0007211218,0.0005120956,0.001824772,0.0006485537,0.001656358,0.002180749,0.0007501585,0.001565326],"category_scores_gemma":[0.01515592,0.0002737911,0.0003610157,0.004264726,0.0007102255,0.004550459,0.00122931,0.0006675829,0.0005267627],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001044106,"about_ca_system_score_gemma":0.001249553,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003013839,"about_ca_topic_score_gemma":0.002981912,"domain_scores_codex":[0.9956512,0.0008073893,0.0004105257,0.0003684578,0.00239332,0.000369194],"domain_scores_gemma":[0.9886262,0.005370529,0.0006638476,0.001962488,0.003041939,0.0003349959],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002192691,0.001094118,0.04704482,0.002193723,0.0003165972,0.0004977164,0.000852946,0.3327611,0.0577754,0.03662123,0.03138889,0.4872608],"study_design_scores_gemma":[0.0001532823,0.002213875,0.009857176,0.0001180854,0.0001088814,0.0004869285,0.0006992775,0.888905,0.0693263,0.01125616,0.01680017,0.00007492142],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8573511,0.004478802,0.1096615,0.0004714524,0.0002255697,0.0003045401,0.003183993,0.005790073,0.01853288],"genre_scores_gemma":[0.9017313,0.001134083,0.08987116,0.0001000565,0.00006467378,0.000168318,0.004607538,0.0003843405,0.001938594],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003013839,"threshold_uncertainty_score":0.01486379,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2301743601","doi":"10.14778/2904483.2904486","title":"Leopard","year":2016,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"VLSI and FPGA Design Techniques","field":"Engineering","cited_by":93,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Graph partition; Computer science; Graph; Space partitioning; Vertex (graph theory); Algorithm; Theoretical computer science","authors":[{"name":"Jiewen Huang","is_ca":false},{"name":"Daniel J. Abadi","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.005755987320689607,"gpt":0.165789516692164,"spread":0.1600335293714744,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004165445,0.0007336564,0.0005107598,0.001020997,0.0009231182,0.001741742,0.001660209,0.0008015404,0.05232141],"category_scores_gemma":[0.001090623,0.0004097563,0.000541923,0.00080649,0.0004008535,0.001635468,0.001724607,0.001022017,0.02448227],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007756401,"about_ca_system_score_gemma":0.0008213957,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003020869,"about_ca_topic_score_gemma":0.005101326,"domain_scores_codex":[0.9994909,0.00004982489,0.00002531981,0.0001331683,0.0002393183,0.00006142884],"domain_scores_gemma":[0.9995421,0.00006643276,0.00002120512,0.0001831965,0.0001367425,0.00005031278],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005969793,0.0001310079,0.001340573,0.0003632906,0.00005307454,0.0004232337,0.0002456598,0.01593347,0.02990586,0.05783994,0.1391491,0.7540178],"study_design_scores_gemma":[0.00008980252,0.0001861433,0.001111761,0.00007617205,0.00002793968,0.0007978191,0.0001244092,0.08816254,0.02427983,0.01729182,0.8677877,0.00006403474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01725239,0.001563006,0.6695555,0.001018135,0.0009159789,0.0004892748,0.002886362,0.05155236,0.254767],"genre_scores_gemma":[0.155492,0.001264074,0.5487882,0.0008712166,0.0001595604,0.0004044976,0.01001806,0.006759912,0.2762425],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.05232141,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}