{"meta":{"query_hash":"ee4a48787628","filters":{"venue":"ACM Transactions on Database Systems"},"cohort_total":43,"direct_labels_cover":0,"predictions_cover":43,"exported":43,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/ee4a48787628","api":"https://metacan.xera.ac/api/v1/cohort?venue=ACM+Transactions+on+Database+Systems"},"results":[{"id":"W1967155922","doi":"10.1145/1670243.1670250","title":"Automatic virtual machine configuration for database workloads","year":2008,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Cloud Computing and Resource Management","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); University of Waterloo","funders":"","keywords":"Computer science; Online transaction processing; Workload; Database; Software deployment; Virtual machine; Virtualization; Temporal isolation among virtual machines; Distributed computing; Hypervisor; Operating system; Cloud computing; Transaction processing; Database transaction","score_opus":0.03302318292325333,"score_gpt":0.2566651177257133,"score_spread":0.22364193480245997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1967155922","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31823432,0.0011372283,0.62627494,0.0003227658,0.00011303036,0.0002929646,0.00070501526,0.04982266,0.003097045],"genre_scores_gemma":[0.78383803,0.00015785276,0.21328862,0.000066779365,0.000041666015,0.00010216808,0.0011313707,0.0007871939,0.0005863814],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99773777,0.0007526432,0.00019313389,0.0005834186,0.0005329253,0.00020002623],"domain_scores_gemma":[0.9959487,0.0016306387,0.00047020378,0.0010915122,0.0007133852,0.000145582],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001503598,0.0013675088,0.0013012233,0.0016960915,0.00088670803,0.0014948723,0.0018698144,0.00061216153,0.0011195217],"category_scores_gemma":[0.010185086,0.00069178804,0.00040143816,0.0013385224,0.00036091075,0.0019201501,0.00069267966,0.00084639876,0.0006737361],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001306497,0.00036698286,0.029353123,0.00022964214,0.000121518744,0.00022162532,0.00044676484,0.15454936,0.06592966,0.0022195394,0.010676124,0.73457927],"study_design_scores_gemma":[0.00005307561,0.000082119084,0.0042087934,0.000015339107,0.000025753958,0.00011372373,0.00009857824,0.9709901,0.020988772,0.0016845436,0.001703653,0.000035610996],"about_ca_topic_score_codex":0.0037357095,"about_ca_topic_score_gemma":0.005414796,"teacher_disagreement_score":0.0037357095,"about_ca_system_score_codex":0.00092193065,"about_ca_system_score_gemma":0.0014175754,"threshold_uncertainty_score":0.007951915},"labels":[],"label_agreement":null},{"id":"W1971512992","doi":"10.1145/1005566.1005570","title":"A compressed accessibility map for XML","year":2004,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Access Control and Trust","field":"Social Sciences","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Efficient XML Interchange; XML validation; XML Schema (W3C); XML database; XML; Streaming XML; XML framework; XML Encryption; Simple API for XML; Database; Document Structure Description; XML Schema Editor; XML Signature; Information retrieval; Data mining; World Wide Web","score_opus":0.05599528690366123,"score_gpt":0.3495931178292262,"score_spread":0.29359783092556496,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1971512992","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020668935,0.00032768672,0.96228725,0.0004524059,0.000111640664,0.00026294132,0.0025536167,0.005315608,0.0080199195],"genre_scores_gemma":[0.34156266,0.000551429,0.6417908,0.00019818301,0.0001570898,0.000691215,0.0061580976,0.0006011294,0.008289333],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9991061,0.00016176124,0.00006736426,0.00015959349,0.00042835803,0.00007668817],"domain_scores_gemma":[0.9975442,0.001030188,0.00012398875,0.00070036686,0.0005135685,0.00008751998],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004691216,0.00043233877,0.00052044186,0.0017303211,0.0008155746,0.0013420627,0.00093761657,0.00061567576,0.007865149],"category_scores_gemma":[0.008138038,0.00026910225,0.00052985764,0.0022853066,0.0007888674,0.0034691424,0.0025046438,0.0009023851,0.0014876882],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050087634,0.00014818872,0.0016814689,0.00040215967,0.00003189439,0.0007170132,0.0008736888,0.09208555,0.014523084,0.2699888,0.041087363,0.5779599],"study_design_scores_gemma":[0.00006280319,0.00016437238,0.0010580972,0.00006391917,0.00002967756,0.0010616309,0.0004321215,0.6672449,0.01726962,0.2246378,0.08790958,0.00006537665],"about_ca_topic_score_codex":0.0038280096,"about_ca_topic_score_gemma":0.0029527447,"teacher_disagreement_score":0.007865149,"about_ca_system_score_codex":0.0007047498,"about_ca_system_score_gemma":0.0010055084,"threshold_uncertainty_score":0.026311576},"labels":[],"label_agreement":null},{"id":"W1971604907","doi":"10.1145/1189769.1189770","title":"Expressive power of an algebra for data mining","year":2006,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Tuple; Relational algebra; Data mining; Data model (GIS); Relational database; Information retrieval; Theoretical computer science; Algebra over a field; Artificial intelligence; Mathematics; Discrete mathematics","score_opus":0.04314760138777388,"score_gpt":0.2949769826758806,"score_spread":0.25182938128810667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1971604907","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013123582,0.0012064434,0.96756905,0.0018655075,0.00011244695,0.00009486209,0.0005946875,0.00053223595,0.01490112],"genre_scores_gemma":[0.34035403,0.001831931,0.64685446,0.001277994,0.0006556034,0.00055812526,0.0016232902,0.0002194492,0.006625156],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99214995,0.0027262531,0.0011550802,0.0012846679,0.002368671,0.0003154164],"domain_scores_gemma":[0.99000376,0.0055053467,0.0006847448,0.0023470656,0.0010431268,0.00041590928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009194762,0.000659631,0.0008335595,0.0017388246,0.0016720496,0.00597976,0.0024840247,0.0013950055,0.0036022456],"category_scores_gemma":[0.012772937,0.00070165,0.002479599,0.0026664382,0.0054518934,0.012588185,0.00388147,0.0038782305,0.0011044861],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002137606,0.000021497495,0.00018275257,0.00007300366,0.000016524284,0.000052398747,0.00026806176,0.001991876,0.0008136031,0.9873696,0.0007066041,0.008482634],"study_design_scores_gemma":[0.000010319063,0.000017573586,0.00005582797,0.000020054205,0.0000104063565,0.00008412787,0.0000491363,0.012067114,0.0005956349,0.9794174,0.00765925,0.000013067641],"about_ca_topic_score_codex":0.0011675557,"about_ca_topic_score_gemma":0.0008031138,"teacher_disagreement_score":0.009194762,"about_ca_system_score_codex":0.00168777,"about_ca_system_score_gemma":0.0017387386,"threshold_uncertainty_score":0.04862714},"labels":[],"label_agreement":null},{"id":"W1973115038","doi":"10.1145/937598.937601","title":"Are quorums an alternative for data replication?","year":2003,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Distributed systems and fault tolerance","field":"Computer Science","cited_by":144,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Replication (statistics); Scalability; Overhead (engineering); Replicate; Distributed computing; Consistency (knowledge bases); Eventual consistency; Selection (genetic algorithm); Data access; Data consistency; Consistency model; Database; Programming language; Artificial intelligence","score_opus":0.1108332601636947,"score_gpt":0.3436451194508417,"score_spread":0.23281185928714698,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1973115038","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021821149,0.02333485,0.8538089,0.02731874,0.006330821,0.00031462993,0.0002766356,0.0009239999,0.06587033],"genre_scores_gemma":[0.59340113,0.012747913,0.32418928,0.007126458,0.0043412824,0.00096669653,0.00040331107,0.000573126,0.05625075],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9887707,0.0046387915,0.0006261876,0.0012117876,0.0041046375,0.0006479465],"domain_scores_gemma":[0.9815678,0.008125682,0.001207764,0.0052660927,0.0031572187,0.00067543326],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011570114,0.0006131641,0.0015698657,0.0013174716,0.0021908467,0.005164957,0.003463542,0.004455815,0.011720119],"category_scores_gemma":[0.030140024,0.00058999326,0.0012551951,0.002286652,0.0036702377,0.019015264,0.002822037,0.002575668,0.004207607],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033022906,0.000081459926,0.0008300834,0.0005075659,0.00006909565,0.0001681942,0.000520527,0.005314398,0.002734002,0.8717052,0.010064313,0.10767491],"study_design_scores_gemma":[0.00017488017,0.00038419265,0.00037047637,0.00026959565,0.00006602512,0.00082316966,0.0005621046,0.032098144,0.0035827623,0.7969105,0.16465642,0.0001016923],"about_ca_topic_score_codex":0.0010642908,"about_ca_topic_score_gemma":0.00073405117,"teacher_disagreement_score":0.011720119,"about_ca_system_score_codex":0.0016708402,"about_ca_system_score_gemma":0.0016125875,"threshold_uncertainty_score":0.061189353},"labels":[],"label_agreement":null},{"id":"W1983304353","doi":"10.1145/974750.974757","title":"A normal form for XML documents","year":2004,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":285,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Functional dependency; XML validation; Document Structure Description; XML; Information retrieval; XML database; XML Schema Editor; Relational database; Theoretical computer science; Programming language; World Wide Web","score_opus":0.02221699491570806,"score_gpt":0.2760328890509727,"score_spread":0.2538158941352646,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1983304353","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037814195,0.0003703671,0.9838071,0.00095297786,0.00037209695,0.00020241308,0.0008956241,0.0016569823,0.007961028],"genre_scores_gemma":[0.071618274,0.0009324224,0.9108495,0.00051492685,0.00034423286,0.000592833,0.0025330936,0.0007806217,0.011834062],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99445254,0.0011183302,0.0012037047,0.0011861986,0.0017997772,0.00023936525],"domain_scores_gemma":[0.99478215,0.0016989767,0.00044212784,0.0013324145,0.0015884856,0.0001559346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040671197,0.0008807774,0.0006523783,0.00217742,0.001047653,0.004094907,0.0011244082,0.0013282045,0.004795743],"category_scores_gemma":[0.010869498,0.00066681317,0.00094288087,0.0022937646,0.0030745335,0.0089872675,0.0020935973,0.003126266,0.0027727343],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047891368,0.000025450003,0.00032219273,0.00014150486,0.000010369835,0.0001340775,0.00053770276,0.0023209485,0.0027929647,0.90567094,0.008598844,0.07939696],"study_design_scores_gemma":[0.000030248097,0.0000650839,0.00015700402,0.00012182789,0.000018772856,0.0007831081,0.00018971885,0.023255708,0.0057443245,0.68327665,0.28631026,0.000047303733],"about_ca_topic_score_codex":0.0019024407,"about_ca_topic_score_gemma":0.0015462369,"teacher_disagreement_score":0.004795743,"about_ca_system_score_codex":0.0017783416,"about_ca_system_score_gemma":0.0018667502,"threshold_uncertainty_score":0.02150923},"labels":[],"label_agreement":null},{"id":"W1984237017","doi":"10.1145/1735886.1735887","title":"Transparent anonymization","year":2010,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Nanyang Technological University; Glaucoma Research Foundation","keywords":"Computer science; Data publishing; Adversary; Exploit; Guard (computer science); Generalization; Data anonymization; Privacy protection; Computer security; Adversary model; Information privacy; Data mining; Publishing","score_opus":0.04600808504767626,"score_gpt":0.28701188294021884,"score_spread":0.2410037978925426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1984237017","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010607074,0.00061348034,0.98037124,0.0008514724,0.00010559414,0.00032593164,0.00026848528,0.00080555724,0.0060511567],"genre_scores_gemma":[0.5811741,0.0017951938,0.40683544,0.0009319015,0.0004923658,0.00079039537,0.00082579145,0.00028446125,0.0068703564],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9805659,0.0076912213,0.0013731478,0.0036949862,0.0056738453,0.0010009222],"domain_scores_gemma":[0.94280505,0.011510584,0.005117999,0.037301015,0.0027045193,0.0005609458],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011675456,0.001101779,0.0017464203,0.002466794,0.0031316618,0.0046039876,0.003558084,0.0023142619,0.0028451507],"category_scores_gemma":[0.034344207,0.000900611,0.002577141,0.0041076955,0.0039072596,0.010458846,0.010193565,0.0038824475,0.0012795585],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036093773,0.0002130696,0.003624159,0.00042171526,0.00027324856,0.0003518387,0.0014130301,0.07484934,0.010205223,0.6811881,0.010299943,0.21679942],"study_design_scores_gemma":[0.00011009191,0.00020464198,0.0012359074,0.00018144163,0.0001438535,0.0020235227,0.00042500877,0.31521332,0.025846737,0.58868873,0.06575921,0.00016756909],"about_ca_topic_score_codex":0.0008325664,"about_ca_topic_score_gemma":0.0006849193,"teacher_disagreement_score":0.011675456,"about_ca_system_score_codex":0.0019928566,"about_ca_system_score_gemma":0.0035242003,"threshold_uncertainty_score":0.061746478},"labels":[],"label_agreement":null},{"id":"W1993831977","doi":"10.1145/1242524.1242529","title":"Estimating the selectivity of approximate string queries","year":2007,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Substring; String (physics); Estimator; Approximate string matching; Computer science; Inverse; String searching algorithm; String metric; Algorithm; String kernel; Data structure; Mathematics; Statistics; Pattern matching; Artificial intelligence","score_opus":0.025787043361268296,"score_gpt":0.2745853299599472,"score_spread":0.24879828659867892,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1993831977","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42199904,0.000864077,0.5711316,0.0005590445,0.00004554847,0.00014139953,0.0011007761,0.0019577881,0.00220072],"genre_scores_gemma":[0.88128126,0.00048353194,0.11513948,0.00014096241,0.00011212254,0.000121461126,0.0018350474,0.00013081155,0.00075532164],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99323285,0.0016519026,0.0006822789,0.0007519881,0.0031247165,0.0005562415],"domain_scores_gemma":[0.96677446,0.023671227,0.002313762,0.0038436044,0.0030561455,0.00034081738],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050851656,0.00050607964,0.001518626,0.0030925204,0.0006027263,0.0021214574,0.0012139954,0.0012437903,0.00093852007],"category_scores_gemma":[0.04692527,0.0004425911,0.00041327867,0.0035054286,0.00088486134,0.004911946,0.0021099458,0.0009016392,0.00054644205],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030186584,0.00030223164,0.111643985,0.00034670712,0.00017960531,0.00066019344,0.00074404513,0.27105168,0.035211578,0.030576503,0.0064391927,0.5398257],"study_design_scores_gemma":[0.000034244313,0.000108372486,0.005065416,0.00001775943,0.000018641369,0.00043627655,0.00024976436,0.96380967,0.015073764,0.013992233,0.001168167,0.000025656904],"about_ca_topic_score_codex":0.0022313942,"about_ca_topic_score_gemma":0.0017145829,"teacher_disagreement_score":0.0050851656,"about_ca_system_score_codex":0.0010693168,"about_ca_system_score_gemma":0.0013338628,"threshold_uncertainty_score":0.026893258},"labels":[],"label_agreement":null},{"id":"W1997533292","doi":"10.1145/503099.503102","title":"SchemaSQL","year":2001,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":94,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; SQL; Data definition language; Relational database; Schema (genetic algorithms); Database schema; Programming language; Database; Conceptual schema; Semi-structured model; Database model; Information retrieval; Database design","score_opus":0.03070971332297829,"score_gpt":0.26929298189477086,"score_spread":0.23858326857179257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1997533292","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027523173,0.0021437951,0.6742128,0.00398384,0.001196718,0.0016733913,0.06256807,0.17475957,0.07670942],"genre_scores_gemma":[0.046233952,0.004812209,0.5766754,0.008545077,0.000719474,0.002589094,0.27213436,0.035607494,0.052682918],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99209315,0.0017938126,0.0013899966,0.0010258692,0.0030110178,0.0006861448],"domain_scores_gemma":[0.99287766,0.0018569377,0.0004363914,0.0022184642,0.002177951,0.00043252858],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007928503,0.0016373199,0.0013677296,0.0028632581,0.001608906,0.008521542,0.0077120685,0.0026352196,0.05195415],"category_scores_gemma":[0.016723096,0.0020172133,0.0036713977,0.003119037,0.001317966,0.010766727,0.007945903,0.0044160215,0.033685356],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005714045,0.00022105538,0.0022787577,0.0019597958,0.00031936404,0.0005276238,0.0008744594,0.0053454535,0.004737686,0.2642291,0.56247634,0.15645906],"study_design_scores_gemma":[0.00010456918,0.00003734294,0.00024640956,0.00017030197,0.00005001328,0.00039460234,0.00016016769,0.01058917,0.0039696572,0.05993186,0.9242697,0.00007622639],"about_ca_topic_score_codex":0.0075289584,"about_ca_topic_score_gemma":0.005942179,"teacher_disagreement_score":0.05195415,"about_ca_system_score_codex":0.0018060249,"about_ca_system_score_gemma":0.0051471824,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W1998597078","doi":"10.1145/1670243.1670248","title":"An information-theoretic analysis of worst-case redundancy in database design","year":2008,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Seventh Framework Programme; Engineering and Physical Sciences Research Council","keywords":"Computer science; Redundancy (engineering); Schema (genetic algorithms); Functional dependency; Database schema; Theoretical computer science; Data integrity; Relational database; Data mining; Database design; Information retrieval; Database","score_opus":0.03518479320693964,"score_gpt":0.27502739543239096,"score_spread":0.23984260222545134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1998597078","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034553234,0.0010741426,0.9562877,0.0009785042,0.00004430057,0.00007382144,0.0001312122,0.00020760063,0.0066494253],"genre_scores_gemma":[0.67349154,0.0015437064,0.32092574,0.0003375761,0.00032304626,0.00038394096,0.00034143857,0.00020104113,0.0024519188],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9830073,0.006897724,0.0011155807,0.0014044441,0.006760076,0.00081481994],"domain_scores_gemma":[0.9477338,0.036994893,0.0034103545,0.0076716146,0.0037428993,0.0004464647],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012243633,0.0012807553,0.0011139945,0.0032944838,0.00094956625,0.0043771984,0.002695252,0.0014210612,0.0023861444],"category_scores_gemma":[0.051508173,0.0010819343,0.0012415366,0.00375176,0.005014334,0.010851688,0.0018904309,0.0024880618,0.0004928748],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001770875,0.00008537819,0.0011470986,0.00032096967,0.00009803643,0.00019272356,0.00040425957,0.3119412,0.004790825,0.62018377,0.0014717194,0.059186947],"study_design_scores_gemma":[0.00002261202,0.00015723029,0.00036575968,0.000054682354,0.000062079926,0.00030502104,0.00007572305,0.41109473,0.004040639,0.5806116,0.0031694162,0.0000405118],"about_ca_topic_score_codex":0.00070240634,"about_ca_topic_score_gemma":0.000611911,"teacher_disagreement_score":0.012243633,"about_ca_system_score_codex":0.003522623,"about_ca_system_score_gemma":0.0015650146,"threshold_uncertainty_score":0.06475127},"labels":[],"label_agreement":null},{"id":"W2002932370","doi":"10.1145/1061318.1061324","title":"Concise descriptions of subsets of structured sets","year":2005,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Automatic summarization; Theoretical computer science; Set (abstract data type); Minimum description length; Online analytical processing; Hierarchy; Cover (algebra); Context (archaeology); Representation (politics); Data mining; Algorithm; Artificial intelligence; Data warehouse","score_opus":0.0319806076430219,"score_gpt":0.26392649524481654,"score_spread":0.23194588760179463,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002932370","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07561506,0.0022776527,0.9070789,0.0015665243,0.000101353624,0.00021281933,0.0035971568,0.00078893604,0.00876161],"genre_scores_gemma":[0.46506375,0.0020317049,0.515063,0.00036826634,0.00021976203,0.0005280953,0.007841733,0.00035356602,0.0085301455],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99712735,0.0009212605,0.00034685564,0.00044658274,0.0009473841,0.00021059053],"domain_scores_gemma":[0.9936215,0.0034808111,0.00086413225,0.0011127458,0.00070964976,0.00021115379],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022118103,0.00064115424,0.0010261646,0.002618103,0.0007449456,0.003211886,0.0017840117,0.00091766665,0.0041692504],"category_scores_gemma":[0.011796633,0.0007064867,0.0010027918,0.0041648815,0.0012099728,0.00964782,0.0019796018,0.001231166,0.0007707657],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021817422,0.00008357257,0.0012826456,0.0006732363,0.00008319858,0.00056547305,0.0014247168,0.075238675,0.0047289445,0.7949336,0.009184256,0.11158349],"study_design_scores_gemma":[0.000047852925,0.00014885595,0.0005138396,0.0001562373,0.00007466788,0.00053334516,0.0007957797,0.2015044,0.005666014,0.7435764,0.046930987,0.000051695726],"about_ca_topic_score_codex":0.0011068818,"about_ca_topic_score_gemma":0.0016557068,"teacher_disagreement_score":0.0041692504,"about_ca_system_score_codex":0.0013782969,"about_ca_system_score_gemma":0.00087727554,"threshold_uncertainty_score":0.0139475465},"labels":[],"label_agreement":null},{"id":"W2006315879","doi":"10.1145/1093382.1093388","title":"Capturing summarizability with integrity constraints in OLAP","year":2005,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":93,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Online analytical processing; Computer science; Heuristics; Dimension (graph theory); Theoretical computer science; Set (abstract data type); Data integrity; Aggregate (composite); Intrinsic dimension; Class (philosophy); Space (punctuation); Data mining; Curse of dimensionality; Data warehouse; Database; Artificial intelligence; Mathematics; Programming language","score_opus":0.025716485551273413,"score_gpt":0.2608755870518358,"score_spread":0.23515910150056235,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2006315879","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040597126,0.00040571202,0.9543814,0.000977661,0.00003952698,0.00028467,0.00058458536,0.0013283198,0.0014010004],"genre_scores_gemma":[0.3083235,0.00037472707,0.68846494,0.00025959924,0.00010451922,0.000260051,0.0012502891,0.00025169863,0.0007107307],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98962605,0.0036393735,0.0010369365,0.0014146612,0.0036239058,0.00065898435],"domain_scores_gemma":[0.9549343,0.030995954,0.0041351155,0.005306885,0.0041178036,0.0005098787],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0058784573,0.0013108379,0.001289405,0.002267622,0.0018945255,0.0046996167,0.003100541,0.002210974,0.0018602543],"category_scores_gemma":[0.056476925,0.0013270716,0.0015704788,0.004584523,0.002596858,0.015227198,0.0045639146,0.0041152826,0.00035042156],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006044139,0.00034751283,0.008910788,0.0008833761,0.00027045608,0.001265886,0.0024500298,0.45767266,0.013429612,0.22160663,0.006796409,0.28576216],"study_design_scores_gemma":[0.000040652074,0.00008267761,0.0007452574,0.00007559753,0.00006857922,0.00026292712,0.00054226496,0.7197083,0.009812639,0.26353425,0.0050631105,0.00006373149],"about_ca_topic_score_codex":0.0074467794,"about_ca_topic_score_gemma":0.0070249117,"teacher_disagreement_score":0.0074467794,"about_ca_system_score_codex":0.0017874858,"about_ca_system_score_gemma":0.0020804398,"threshold_uncertainty_score":0.03108865},"labels":[],"label_agreement":null},{"id":"W2010916971","doi":"10.1145/1132863.1132868","title":"Integrating XML data sources using approximate joins","year":2006,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Joins; XML validation; XML; Efficient XML Interchange; Document Structure Description; XML Schema (W3C); XML database; Information retrieval; Streaming XML; Data mining; Set (abstract data type); Tree (set theory); XML Encryption; Database; Programming language; World Wide Web","score_opus":0.06748389564404962,"score_gpt":0.2828255905583581,"score_spread":0.2153416949143085,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2010916971","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014708839,0.00036875784,0.98290294,0.0001556621,0.00002253231,0.000094855524,0.00015531245,0.0009198168,0.0006712509],"genre_scores_gemma":[0.15698422,0.0004293297,0.84011525,0.00012063934,0.000067847206,0.0001603939,0.0011033483,0.00022004689,0.00079888257],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9802822,0.0062700724,0.0015100472,0.0030749722,0.00837563,0.00048711346],"domain_scores_gemma":[0.977114,0.011195478,0.0025240916,0.006072334,0.0026196546,0.000474393],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013593591,0.0012680612,0.0023426497,0.0047071683,0.001813167,0.007316389,0.0036772161,0.0019972145,0.0011685237],"category_scores_gemma":[0.042752165,0.0012276745,0.0020082414,0.010205037,0.0016512673,0.011064926,0.00909385,0.002113866,0.0008928873],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000540636,0.00026710518,0.0132048745,0.00041290987,0.00042221908,0.0007013024,0.002189641,0.34809458,0.015415243,0.15034911,0.0043847375,0.46401763],"study_design_scores_gemma":[0.0000320051,0.00014600491,0.0008764026,0.000044096498,0.000064311345,0.00036835135,0.0004017719,0.903037,0.0081827715,0.08020989,0.006586715,0.000050680563],"about_ca_topic_score_codex":0.003814023,"about_ca_topic_score_gemma":0.003764135,"teacher_disagreement_score":0.013593591,"about_ca_system_score_codex":0.001571089,"about_ca_system_score_gemma":0.0025510434,"threshold_uncertainty_score":0.07189059},"labels":[],"label_agreement":null},{"id":"W2036787492","doi":"10.1145/2560796","title":"Sharing across Multiple MapReduce Jobs","year":2014,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Cloud Computing and Resource Management","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Cloud computing; Merge (version control); Key (lock); Batch processing; Database; Distributed computing; Context (archaeology); Parallel computing; Operating system","score_opus":0.028705162256593777,"score_gpt":0.26509984859053803,"score_spread":0.23639468633394425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036787492","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18690397,0.0006699495,0.7822013,0.0010932782,0.00035198164,0.0009878745,0.0007079911,0.007930519,0.019153155],"genre_scores_gemma":[0.800671,0.00021367273,0.1880477,0.00028210235,0.00015997076,0.000378338,0.0012259005,0.00069861097,0.0083228415],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9951001,0.0007445594,0.0003291403,0.0012356312,0.0017532642,0.00083726156],"domain_scores_gemma":[0.9959973,0.0004879944,0.00016136361,0.0020764272,0.00080448884,0.00047239347],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029245259,0.0013915529,0.0016991477,0.00077832333,0.002184439,0.0028343822,0.0050093117,0.00085879257,0.0034212682],"category_scores_gemma":[0.0048481096,0.00090139953,0.0018375341,0.001324777,0.0009719788,0.0035215355,0.004795258,0.0013468964,0.0014923448],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023405585,0.0016803972,0.013521366,0.00070483744,0.0009671646,0.0014960166,0.0019489669,0.37149304,0.08391492,0.03782113,0.030838398,0.45327312],"study_design_scores_gemma":[0.00015564363,0.00046571702,0.004621177,0.00003093931,0.00017492493,0.0005285574,0.000927632,0.87582636,0.041220687,0.047517773,0.028372929,0.00015771548],"about_ca_topic_score_codex":0.0061021615,"about_ca_topic_score_gemma":0.005765615,"teacher_disagreement_score":0.0061021615,"about_ca_system_score_codex":0.0013213747,"about_ca_system_score_gemma":0.003646484,"threshold_uncertainty_score":0.015466571},"labels":[],"label_agreement":null},{"id":"W2042315498","doi":"10.1145/581751.581752","title":"Searching for dependencies at multiple abstraction levels","year":2002,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Functional dependency; Dependency (UML); Tuple; Online analytical processing; Dependency theory (database theory); Generalization; Abstraction; Scope (computer science); Data mining; Extension (predicate logic); Theoretical computer science; Database; Data warehouse; Relational database; Artificial intelligence; Programming language","score_opus":0.09694704259971397,"score_gpt":0.2879969122087368,"score_spread":0.19104986960902282,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2042315498","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29665953,0.0005798722,0.6925,0.00059160957,0.000033789227,0.00018875186,0.0018381963,0.0038118928,0.0037963416],"genre_scores_gemma":[0.7284258,0.0002298157,0.26670104,0.00011344503,0.00002280193,0.000087460445,0.0025494357,0.00019027208,0.00168001],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99852175,0.00018844106,0.00011762486,0.00036820048,0.0005274123,0.00027655886],"domain_scores_gemma":[0.99395156,0.0032364035,0.00062897796,0.0010570455,0.00086342485,0.00026266693],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014630222,0.0006634902,0.0007590489,0.0031319268,0.0012602,0.0014684391,0.0011683278,0.0008536466,0.002637543],"category_scores_gemma":[0.009311095,0.00078402204,0.0012519724,0.0017216491,0.0006713869,0.0038950837,0.0022272577,0.0014495861,0.0005226152],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008020498,0.00044229496,0.11373778,0.0008394884,0.00031726874,0.0032343115,0.0023241572,0.09193343,0.039309334,0.07596149,0.0128951315,0.6582033],"study_design_scores_gemma":[0.000044965298,0.0003082459,0.029141728,0.00017063605,0.0003204052,0.001496567,0.00105286,0.71309793,0.035868064,0.2005826,0.017787384,0.0001286964],"about_ca_topic_score_codex":0.004696145,"about_ca_topic_score_gemma":0.0077070603,"teacher_disagreement_score":0.004696145,"about_ca_system_score_codex":0.00069668697,"about_ca_system_score_gemma":0.0014138488,"threshold_uncertainty_score":0.009337604},"labels":[],"label_agreement":null},{"id":"W2051445745","doi":"10.1145/507234.507237","title":"A logical foundation for deductive object-oriented databases","year":2002,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Logic, Reasoning, and Knowledge","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Programming language; Deductive database; Syntax; Semantics (computer science); Class (philosophy); Object (grammar); Well-founded semantics; Logic programming; Inheritance (genetic algorithm); Operational semantics; Database; Natural language processing; Artificial intelligence; Denotational semantics","score_opus":0.07233550198391613,"score_gpt":0.2909125887188853,"score_spread":0.21857708673496917,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2051445745","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004130984,0.0034134015,0.96950567,0.006341223,0.00046727713,0.00019217293,0.00030180844,0.00065475336,0.014992857],"genre_scores_gemma":[0.1374639,0.005820825,0.8409525,0.0044378513,0.0015618816,0.00063276873,0.0009320486,0.0003185922,0.007879667],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9940128,0.0014617819,0.0007951354,0.0008406765,0.0025058312,0.00038382344],"domain_scores_gemma":[0.9916986,0.00417873,0.00067912106,0.0010404062,0.0019477073,0.00045545422],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008905284,0.0007730386,0.0010224988,0.003470689,0.0032393057,0.009901508,0.0031213006,0.0029694694,0.0039972183],"category_scores_gemma":[0.012802687,0.0014597184,0.0022581718,0.003078677,0.010634559,0.019101996,0.0047744177,0.0063579073,0.0020400109],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000005690585,0.000020125452,0.00007391792,0.000057148653,0.0000061043615,0.00008243231,0.0002636209,0.00054763386,0.00020572329,0.9920258,0.0011258293,0.005585916],"study_design_scores_gemma":[0.000020238629,0.000015966694,0.000048981856,0.000061405204,0.000010242909,0.00010758383,0.00008008367,0.003917661,0.00038446137,0.9625782,0.03276023,0.0000149607495],"about_ca_topic_score_codex":0.0022547739,"about_ca_topic_score_gemma":0.0013230507,"teacher_disagreement_score":0.009901508,"about_ca_system_score_codex":0.0029527324,"about_ca_system_score_gemma":0.003697742,"threshold_uncertainty_score":0.047096193},"labels":[],"label_agreement":null},{"id":"W2053495332","doi":"10.1145/2487259.2487260","title":"Analysis and optimization for boolean expression indexing","year":2013,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Regular expression; Theoretical computer science; Search engine indexing; Boolean expression; Tree (set theory); Matching (statistics); Search tree; Data structure; Tree structure; String searching algorithm; Data mining; Algorithm; Pattern matching; Boolean function; Binary tree; Search algorithm; Artificial intelligence; Mathematics","score_opus":0.020552758876912467,"score_gpt":0.24775622824136947,"score_spread":0.227203469364457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053495332","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030712394,0.0013236859,0.95371044,0.00058461766,0.00009096297,0.00016407442,0.0006802697,0.0023304906,0.010403055],"genre_scores_gemma":[0.3858693,0.0011505982,0.60258126,0.0003091113,0.00015022229,0.000394422,0.0021345243,0.0007629616,0.006647552],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99615735,0.00070424593,0.00027502654,0.0003671852,0.0020358574,0.0004603751],"domain_scores_gemma":[0.9951615,0.0026894722,0.0003229961,0.00093716744,0.00078330684,0.000105482526],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017930187,0.0007183676,0.001065706,0.0014823194,0.00075101183,0.0030088068,0.0019651477,0.00067935744,0.0049359775],"category_scores_gemma":[0.011212532,0.000394244,0.00097509404,0.004083353,0.0011632888,0.0056216894,0.0016661857,0.0014642633,0.0011391459],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005240041,0.0002933208,0.0031861453,0.000523604,0.00007034847,0.00012318243,0.00022885909,0.25457618,0.013799619,0.24985844,0.016724605,0.46009165],"study_design_scores_gemma":[0.000030232175,0.000087155444,0.0003147385,0.000023014301,0.000024287303,0.000077240846,0.000061566265,0.9092592,0.0061206357,0.07774493,0.0062395153,0.000017521454],"about_ca_topic_score_codex":0.0035463027,"about_ca_topic_score_gemma":0.005173323,"teacher_disagreement_score":0.0049359775,"about_ca_system_score_codex":0.0024029857,"about_ca_system_score_gemma":0.0032632498,"threshold_uncertainty_score":0.017435014},"labels":[],"label_agreement":null},{"id":"W2054655669","doi":"10.1145/383891.383892","title":"Querying ATSQL databases with temporal logic","year":2001,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; SQL; Temporal logic; Expressive power; Temporal database; Extension (predicate logic); Database; Translation (biology); Programming language; Representation (politics); Theoretical computer science","score_opus":0.05051783123629092,"score_gpt":0.2832165114231688,"score_spread":0.23269868018687787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2054655669","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021816252,0.00018842761,0.9672636,0.00061631174,0.00007785459,0.00009233407,0.000753834,0.003816975,0.0053743753],"genre_scores_gemma":[0.47757867,0.0007945401,0.5091717,0.0010111053,0.00031091797,0.00036441867,0.0039546015,0.00089132594,0.005922742],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9956501,0.0011407754,0.00069717254,0.0004340682,0.0018463063,0.00023146595],"domain_scores_gemma":[0.9949386,0.002177153,0.0005139877,0.001131729,0.0011009211,0.00013755982],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002744949,0.0005990789,0.00058518205,0.00084578927,0.00065524905,0.0031168598,0.0015705588,0.00062208076,0.0031290401],"category_scores_gemma":[0.007968815,0.0004623743,0.0010913615,0.0013556501,0.00089986477,0.0062528457,0.0022128979,0.0015121073,0.00066832744],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008290342,0.00024559663,0.0031128346,0.0006475863,0.00015044239,0.0012511491,0.0012773202,0.05012334,0.04116383,0.71939915,0.017952897,0.16384676],"study_design_scores_gemma":[0.0001269976,0.00021442222,0.0005331871,0.00014452344,0.00014878891,0.001196719,0.0004936118,0.5247963,0.05065151,0.33673665,0.08487537,0.00008191195],"about_ca_topic_score_codex":0.0020707618,"about_ca_topic_score_gemma":0.0017258255,"teacher_disagreement_score":0.0031290401,"about_ca_system_score_codex":0.00072468835,"about_ca_system_score_gemma":0.0010716353,"threshold_uncertainty_score":0.01451683},"labels":[],"label_agreement":null},{"id":"W2057058417","doi":"10.1145/1132863.1132873","title":"Approximation and streaming algorithms for histogram construction problems","year":2006,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":127,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Histogram; Computer science; Algorithm; Approximation algorithm; Artificial intelligence; Image (mathematics)","score_opus":0.02572179603381434,"score_gpt":0.24387751952964024,"score_spread":0.2181557234958259,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2057058417","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004234343,0.00071062933,0.99241626,0.00031790038,0.00007288921,0.00008143284,0.00015684994,0.0006124793,0.0013972052],"genre_scores_gemma":[0.12968315,0.0016713652,0.8623673,0.00024870457,0.0003357001,0.00039395443,0.001289884,0.00035428876,0.0036556663],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9975446,0.00072869484,0.00018039197,0.0005269393,0.00072234665,0.00029711332],"domain_scores_gemma":[0.99103725,0.006141248,0.00052085787,0.0012659122,0.0007874742,0.00024717365],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032294055,0.0017991414,0.0018958513,0.0018293496,0.001014513,0.0025607406,0.0032665161,0.0019017508,0.00675954],"category_scores_gemma":[0.021510726,0.00096067495,0.0015089521,0.0043921946,0.0012676676,0.006773975,0.0028024577,0.0040849163,0.0019205038],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004619743,0.0002885616,0.0020136435,0.00067259505,0.00011951982,0.00011534751,0.00042252964,0.44497287,0.0026408017,0.15888882,0.018935662,0.37046757],"study_design_scores_gemma":[0.000051542167,0.000048812544,0.00014980204,0.000025965246,0.000019187959,0.00008077947,0.00006233665,0.88275576,0.0008467631,0.112909555,0.0030343975,0.00001505862],"about_ca_topic_score_codex":0.004042483,"about_ca_topic_score_gemma":0.003596257,"teacher_disagreement_score":0.00675954,"about_ca_system_score_codex":0.0021763283,"about_ca_system_score_gemma":0.0017730151,"threshold_uncertainty_score":0.022612989},"labels":[],"label_agreement":null},{"id":"W2057545667","doi":"10.1145/1508857.1508858","title":"The design of a query monitoring system","year":2009,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":169,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Query optimization; Online aggregation; Joins; Query expansion; Sargable; Query plan; View; Data mining; Database; Query language; Cardinality (data modeling); Web query classification; Information retrieval; Web search query; Search engine; Database design","score_opus":0.034277973946782773,"score_gpt":0.26691792212526705,"score_spread":0.23263994817848427,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2057545667","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010724148,0.00022156566,0.9742353,0.0004908641,0.00009500238,0.0007574908,0.00012286732,0.011996241,0.0013564616],"genre_scores_gemma":[0.21831886,0.00030868078,0.7740512,0.00064785377,0.00023564004,0.0011814734,0.0005507702,0.00077329774,0.003932154],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9951591,0.00079410285,0.0006581429,0.0013716399,0.001706362,0.0003105866],"domain_scores_gemma":[0.99472284,0.0015220954,0.0004387884,0.0010661832,0.0017347932,0.00051537046],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005352603,0.0007736016,0.0013875815,0.0011321045,0.0015894073,0.0043182536,0.0039578876,0.0020913412,0.0033822078],"category_scores_gemma":[0.009265824,0.001312127,0.0006248969,0.0010759275,0.0012107971,0.004988434,0.0026866384,0.0023400155,0.002229899],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034331058,0.0015054166,0.020414403,0.001449778,0.00043143556,0.0017917681,0.0029557608,0.07104931,0.23895007,0.10105241,0.03948348,0.5174831],"study_design_scores_gemma":[0.00028401287,0.0008010718,0.002684423,0.00010414427,0.00025679255,0.0011866451,0.00029632973,0.8149929,0.09537934,0.014746041,0.06909358,0.00017481264],"about_ca_topic_score_codex":0.0019670192,"about_ca_topic_score_gemma":0.00094351225,"teacher_disagreement_score":0.005352603,"about_ca_system_score_codex":0.0009907882,"about_ca_system_score_gemma":0.0025122664,"threshold_uncertainty_score":0.028307617},"labels":[],"label_agreement":null},{"id":"W2080555007","doi":"10.1145/1806907.1806909","title":"Continuous online index tuning in moving object databases","year":2010,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Media Development Authority - Singapore","keywords":"Computer science; Granularity; B-tree; Overhead (engineering); Search engine indexing; Tree (set theory); Grid; Data mining; Workload; Set (abstract data type); Object (grammar); Database; Binary tree; Information retrieval; Algorithm; Artificial intelligence; Mathematics","score_opus":0.028133624355503567,"score_gpt":0.2745241931610565,"score_spread":0.2463905688055529,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2080555007","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24833494,0.0062585506,0.7087886,0.00050114864,0.00038419047,0.0004225628,0.0010084196,0.026139518,0.008162039],"genre_scores_gemma":[0.7060131,0.0009675279,0.28805995,0.00026830946,0.00017810205,0.00021605396,0.0016059786,0.0008411545,0.0018499182],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971294,0.00042051068,0.00030966444,0.0006414478,0.0012082311,0.00029071196],"domain_scores_gemma":[0.99513704,0.0015299259,0.00042669725,0.0017809386,0.0007201069,0.00040533356],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027643365,0.0008555252,0.0014576606,0.0017399283,0.001170492,0.0029372005,0.0039886883,0.00096496625,0.001234371],"category_scores_gemma":[0.011690541,0.0008134115,0.0004406682,0.003571848,0.0008167914,0.0057842457,0.0032309145,0.0010101733,0.0007534404],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018613513,0.0006953197,0.015519371,0.0005334949,0.00019378134,0.0006266456,0.0008018925,0.15864287,0.055576365,0.0150023,0.02525389,0.7252927],"study_design_scores_gemma":[0.00012576205,0.00024497235,0.0029391705,0.000031523647,0.00004676721,0.00037723486,0.0002107677,0.9523668,0.016977448,0.013697742,0.012916772,0.00006496484],"about_ca_topic_score_codex":0.004184304,"about_ca_topic_score_gemma":0.0037411188,"teacher_disagreement_score":0.004184304,"about_ca_system_score_codex":0.0010790012,"about_ca_system_score_gemma":0.0013291815,"threshold_uncertainty_score":0.01461935},"labels":[],"label_agreement":null},{"id":"W2090154982","doi":"10.1145/357775.357778","title":"Emancipating instances from the tyranny of classes in information modeling","year":2000,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":165,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; Memorial University of Newfoundland","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Schema (genetic algorithms); Interoperability; Relational database; Class (philosophy); Semi-structured model; Theoretical computer science; Information schema; Data mining; Information retrieval; Artificial intelligence; Database model; World Wide Web","score_opus":0.0253603979568016,"score_gpt":0.2532512764296321,"score_spread":0.2278908784728305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2090154982","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01777291,0.008733952,0.78210175,0.106151566,0.0013580264,0.00006673618,0.00011881284,0.0003500306,0.083346196],"genre_scores_gemma":[0.5990382,0.010864572,0.35529315,0.015359718,0.0028624907,0.0005916795,0.00023724408,0.00063256756,0.015120401],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9847799,0.008661253,0.0008816159,0.0019234713,0.003014071,0.00073969853],"domain_scores_gemma":[0.9756977,0.014712992,0.0011553179,0.006333902,0.0013865614,0.0007134534],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01802305,0.00082204404,0.0010823683,0.0025143584,0.0036317965,0.0139512755,0.0026584577,0.0056157904,0.0016046552],"category_scores_gemma":[0.02406001,0.0008786718,0.0014480575,0.0031510482,0.039993867,0.02947354,0.01135887,0.01372405,0.0007311671],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000006006442,0.000003904168,0.00014207815,0.000015737502,0.0000037507361,0.000020069823,0.0013238929,0.00035402476,0.0000392839,0.9928981,0.000811284,0.0043818406],"study_design_scores_gemma":[0.000005915294,0.000006610186,0.00006721102,0.00006644332,0.000006217553,0.00004872068,0.00033425668,0.0030968385,0.00018000379,0.9585594,0.037616678,0.000011797189],"about_ca_topic_score_codex":0.006898737,"about_ca_topic_score_gemma":0.0038499339,"teacher_disagreement_score":0.01802305,"about_ca_system_score_codex":0.0052975686,"about_ca_system_score_gemma":0.0039129388,"threshold_uncertainty_score":0.09531611},"labels":[],"label_agreement":null},{"id":"W2104120015","doi":"10.1145/1538909.1538910","title":"Anonymization-based attacks in privacy-preserving data publishing","year":2009,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Research Grants Council, University Grants Committee","keywords":"Computer science; Adversary; Data publishing; Overhead (engineering); Confidentiality; Computer security; Information loss; k-anonymity; Information sensitivity; Publishing; Information privacy; Internet privacy; Artificial intelligence","score_opus":0.07597995798891426,"score_gpt":0.3100199144698071,"score_spread":0.23403995648089282,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104120015","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054147135,0.0010526059,0.93594885,0.0031657452,0.00010466744,0.00025535843,0.00012590086,0.00064878,0.0045510475],"genre_scores_gemma":[0.8135076,0.0006547622,0.18282366,0.0006502374,0.0002000567,0.00024622394,0.00016151456,0.00008718859,0.0016687051],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9641204,0.020603426,0.0024207314,0.0033838896,0.007822008,0.0016495552],"domain_scores_gemma":[0.9185688,0.0446494,0.0062259478,0.027800487,0.0018898209,0.00086563325],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017056737,0.0008382078,0.0021234036,0.0019975204,0.003471215,0.0058100824,0.0036946693,0.004506863,0.0009468051],"category_scores_gemma":[0.04646838,0.0012581425,0.002080936,0.003278606,0.0060535357,0.015711414,0.009024018,0.004878683,0.00041089597],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012460324,0.00032272836,0.006576279,0.00047385387,0.00038018898,0.0008553498,0.0023169583,0.13499816,0.010663788,0.6992227,0.0039061904,0.13903782],"study_design_scores_gemma":[0.0001119433,0.0002781649,0.0005268621,0.00011436007,0.00012339132,0.0017917318,0.00038565104,0.43353227,0.026131188,0.52238613,0.014531484,0.000086818945],"about_ca_topic_score_codex":0.00046336974,"about_ca_topic_score_gemma":0.00027819935,"teacher_disagreement_score":0.017056737,"about_ca_system_score_codex":0.0022071137,"about_ca_system_score_gemma":0.002576445,"threshold_uncertainty_score":0.09020573},"labels":[],"label_agreement":null},{"id":"W2112527958","doi":"10.1145/1538909.1538913","title":"Snapshot isolation and integrity constraints in replicated databases","year":2009,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Distributed systems and fault tolerance","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Seventh Framework Programme; Federación Española de Enfermedades Raras","keywords":"Computer science; Serializability; Correctness; Replica; Data integrity; Snapshot (computer storage); Database; Isolation (microbiology); Distributed computing; Rollback; Concurrency control; Two-phase locking; Distributed database; Fault tolerance; Database transaction; Distributed transaction; Transaction processing; Programming language","score_opus":0.044649816998303886,"score_gpt":0.2969985777953138,"score_spread":0.2523487607970099,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2112527958","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024127575,0.0026730928,0.9545694,0.001631581,0.0002090422,0.0001835933,0.0003711981,0.00041346528,0.015820982],"genre_scores_gemma":[0.6112283,0.0033581103,0.3743851,0.00088348374,0.00070108235,0.0006294785,0.0007677842,0.00021962126,0.007826959],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.992069,0.0023917675,0.0010315495,0.0010830044,0.0028928232,0.0005319468],"domain_scores_gemma":[0.9878407,0.0066860113,0.0013868647,0.002240202,0.0014999803,0.00034620124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047610393,0.0006814023,0.00075693434,0.0012840084,0.001154748,0.0039399555,0.002604337,0.0018812921,0.0019173569],"category_scores_gemma":[0.015123383,0.00077477284,0.0009741287,0.0023605905,0.0046763173,0.00833667,0.0031356614,0.0031787993,0.00049472536],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043698208,0.000038343132,0.00038721794,0.00017862095,0.000022750792,0.0005031926,0.0005667345,0.012771364,0.0025308994,0.965076,0.0013427258,0.016538434],"study_design_scores_gemma":[0.00007298148,0.00009162417,0.0004016838,0.0001195807,0.00005377669,0.00075621164,0.00028607185,0.076322764,0.008460076,0.87642455,0.03695431,0.000056410612],"about_ca_topic_score_codex":0.0020522508,"about_ca_topic_score_gemma":0.0013191352,"teacher_disagreement_score":0.0047610393,"about_ca_system_score_codex":0.0013150302,"about_ca_system_score_gemma":0.0019858188,"threshold_uncertainty_score":0.025179088},"labels":[],"label_agreement":null},{"id":"W2118265701","doi":"10.1145/1189769.1189772","title":"Adaptive rank-aware query optimization in relational databases","year":2006,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Query optimization; Query plan; Ranking (information retrieval); Rank (graph theory); Cardinality (data modeling); Key (lock); Query language; Relational database; Join (topology); Operator (biology); Sargable; Theoretical computer science; Database; Data mining; Information retrieval; Web search query; Search engine","score_opus":0.038032297023068214,"score_gpt":0.24930291190045173,"score_spread":0.21127061487738352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118265701","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09871926,0.0015516743,0.89006954,0.0003835795,0.000033055694,0.00016329435,0.00028995713,0.0065246425,0.002264976],"genre_scores_gemma":[0.58750033,0.00048960384,0.40963784,0.00016390841,0.00004255056,0.000085222906,0.0004544175,0.00041090138,0.0012152055],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9965239,0.00087426836,0.00024337384,0.0006374824,0.0013735977,0.0003472264],"domain_scores_gemma":[0.99738175,0.0012261283,0.0002494766,0.0006344712,0.0003996836,0.00010849378],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002333811,0.0010045992,0.0010021677,0.00078645727,0.0008179749,0.0016426853,0.0021251491,0.0007602485,0.00078309025],"category_scores_gemma":[0.004435072,0.0004756686,0.00074252696,0.0014474225,0.00092571165,0.0027237092,0.0015790609,0.0012279664,0.00032237874],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007820624,0.0004634802,0.0048808963,0.00030347417,0.00013401009,0.00023676112,0.00047147987,0.68791324,0.050939184,0.022223985,0.0059607774,0.2256907],"study_design_scores_gemma":[0.00002664925,0.000098046694,0.00046305632,0.0000050590033,0.00002669086,0.000069724454,0.00006931829,0.98139703,0.009056041,0.007243744,0.0015229634,0.000021618958],"about_ca_topic_score_codex":0.009449322,"about_ca_topic_score_gemma":0.010027917,"teacher_disagreement_score":0.009449322,"about_ca_system_score_codex":0.0011179572,"about_ca_system_score_gemma":0.0017901848,"threshold_uncertainty_score":0.018788636},"labels":[],"label_agreement":null},{"id":"W2121433213","doi":"10.1145/1386118.1386119","title":"Probabilistic top- <i>k</i> and ranking-aggregate queries","year":2008,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":87,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Ranking (information retrieval); Probabilistic logic; Aggregate (composite); Uncertain data; Semantics (computer science); Tuple; Probabilistic database; Search engine indexing; Materialized view; Dimension (graph theory); Information retrieval; Online aggregation; Data mining; Theoretical computer science; Web search query; Sargable; Search engine; Relational database; Artificial intelligence; View; Database theory; Programming language","score_opus":0.027337698147204898,"score_gpt":0.23315272567843237,"score_spread":0.20581502753122746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121433213","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008490557,0.00046477557,0.9854857,0.0006847719,0.00005451708,0.0002391189,0.000633522,0.0010550366,0.0028919748],"genre_scores_gemma":[0.26549426,0.0008102531,0.72831506,0.0004174823,0.00023394081,0.0003359421,0.0012919975,0.0002454158,0.0028556823],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99116546,0.0022345618,0.0012521595,0.0014616789,0.0033079342,0.00057830644],"domain_scores_gemma":[0.98819566,0.004778282,0.0016434215,0.003834311,0.0012288284,0.00031946346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068583973,0.001092663,0.0021339983,0.00230493,0.0017601036,0.006600272,0.003480271,0.0021810532,0.0042334986],"category_scores_gemma":[0.02158022,0.00096074044,0.0022894866,0.005715603,0.0022658166,0.014321743,0.0036583706,0.0024149066,0.0013850861],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052433397,0.00035198792,0.0049428917,0.0010320343,0.00021796717,0.000429306,0.0010223333,0.15808032,0.0063776327,0.5143227,0.020195452,0.2925031],"study_design_scores_gemma":[0.00005626035,0.00014416914,0.0008622931,0.00006395539,0.00008845164,0.0008789226,0.00040758812,0.60600805,0.005140896,0.37075728,0.015475266,0.00011685164],"about_ca_topic_score_codex":0.0029567287,"about_ca_topic_score_gemma":0.004636779,"teacher_disagreement_score":0.0068583973,"about_ca_system_score_codex":0.0015839745,"about_ca_system_score_gemma":0.002713371,"threshold_uncertainty_score":0.036271095},"labels":[],"label_agreement":null},{"id":"W2130696419","doi":"10.1145/1189769.1189774","title":"Towards multidimensional subspace skyline analysis","year":2006,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":117,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Skyline; Linear subspace; Subspace topology; Computer science; Semantics (computer science); Theoretical computer science; Space (punctuation); Object (grammar); Online analytical processing; Algorithm; Data mining; Mathematics; Artificial intelligence; Pure mathematics; Programming language","score_opus":0.017999956148877616,"score_gpt":0.25174250827514694,"score_spread":0.23374255212626932,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2130696419","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0061091236,0.00028143433,0.9918046,0.00013727398,0.000017352171,0.000038139715,0.00024656282,0.0003751121,0.0009903526],"genre_scores_gemma":[0.22323057,0.00092459214,0.7713571,0.00018223272,0.00014793516,0.0002849384,0.0017450209,0.00032944404,0.0017982549],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986136,0.00048369524,0.00008812942,0.00024077842,0.0004206132,0.0001531998],"domain_scores_gemma":[0.99735594,0.0010461592,0.00036124294,0.00043484257,0.00066338974,0.000138462],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014808063,0.0011514407,0.0014396845,0.0028712703,0.00089652237,0.0028612346,0.0012787916,0.0010370555,0.003505152],"category_scores_gemma":[0.0062027206,0.0005003476,0.0017621464,0.004168312,0.0014085002,0.004678326,0.0028762291,0.0025467372,0.0009227031],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002268368,0.000094208015,0.004138349,0.00035189788,0.00014873958,0.00027018166,0.00074235053,0.41420624,0.0055502406,0.4169173,0.009528934,0.14782473],"study_design_scores_gemma":[0.00001043261,0.000029522083,0.00030350307,0.000025538762,0.000010923683,0.00003888863,0.00013771078,0.8286785,0.0009101667,0.16532436,0.0045128944,0.000017546725],"about_ca_topic_score_codex":0.0046133236,"about_ca_topic_score_gemma":0.002814696,"teacher_disagreement_score":0.0046133236,"about_ca_system_score_codex":0.0012027251,"about_ca_system_score_gemma":0.0012347426,"threshold_uncertainty_score":0.011725843},"labels":[],"label_agreement":null},{"id":"W2139000699","doi":"10.1145/2338626.2338633","title":"Reordering rows for better compression","year":2012,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Speedup; Lexicographical order; Row; Sorting; Compression (physics); Algorithm; Data compression; Heuristic; Heuristics; Huffman coding; Compression ratio; Encoding (memory); Parallel computing; Mathematics; Combinatorics; Artificial intelligence; Database","score_opus":0.040661275768922964,"score_gpt":0.289049769394661,"score_spread":0.248388493625738,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2139000699","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17265767,0.0040815184,0.774421,0.0033137163,0.0018193946,0.0005603269,0.0027996476,0.019016147,0.021330627],"genre_scores_gemma":[0.25103354,0.0012680349,0.72864544,0.0014535153,0.00038938542,0.00020062919,0.003628042,0.0025716906,0.010809714],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986203,0.00022235324,0.00021742894,0.00023039417,0.0005053591,0.00020429963],"domain_scores_gemma":[0.9934715,0.0017409571,0.0004224986,0.0030348732,0.0011987454,0.00013142767],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008232598,0.0012797078,0.0010414048,0.0014783329,0.00066859025,0.0023784053,0.0013174145,0.00090730935,0.014385993],"category_scores_gemma":[0.0078937765,0.0005491061,0.00067622174,0.0029803088,0.0007031323,0.004133627,0.0012465388,0.001414162,0.0070073716],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010022202,0.00035039833,0.003524898,0.00072837947,0.00011090166,0.00056430185,0.0004085753,0.02535396,0.19345893,0.025155531,0.04149568,0.7078462],"study_design_scores_gemma":[0.00032111246,0.0009703787,0.0035968137,0.0003803832,0.00023958011,0.002404997,0.00080682826,0.2302873,0.56036377,0.05274783,0.14765249,0.00022857735],"about_ca_topic_score_codex":0.0010832476,"about_ca_topic_score_gemma":0.002479769,"teacher_disagreement_score":0.014385993,"about_ca_system_score_codex":0.0005860187,"about_ca_system_score_gemma":0.0014985465,"threshold_uncertainty_score":0.048125982},"labels":[],"label_agreement":null},{"id":"W2150433972","doi":"10.1145/958942.958944","title":"Efficient dynamic mining of constrained frequent sets","year":2003,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":85,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba; University of British Columbia","funders":"","keywords":"Computer science; Context (archaeology); Computation; Exploit; Process (computing); Set (abstract data type); Data mining; Function (biology); Component (thermodynamics); Association rule learning; Algorithm","score_opus":0.02206422575661429,"score_gpt":0.26971647157293405,"score_spread":0.24765224581631975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2150433972","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04971014,0.00038001023,0.94725746,0.00022965914,0.000023387862,0.00015698417,0.00048364908,0.0009351385,0.00082351547],"genre_scores_gemma":[0.29939854,0.00032274693,0.6969597,0.00008380792,0.000056314508,0.00027906537,0.0021218404,0.000117662246,0.00066037953],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976305,0.00054825394,0.00020891384,0.000454629,0.0009604961,0.00019722624],"domain_scores_gemma":[0.988818,0.007849645,0.0007416127,0.0011988475,0.0011973148,0.00019469649],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020639924,0.00074351515,0.0015004713,0.003555853,0.0008334368,0.0014380915,0.0020157585,0.00087687024,0.0012586927],"category_scores_gemma":[0.020683946,0.00068254385,0.0008662645,0.0041895485,0.00055737305,0.0030653367,0.0015723222,0.0009170439,0.0004948013],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00056158117,0.00022768886,0.008264381,0.00041677148,0.00018451946,0.00068065216,0.0004820435,0.19733585,0.012072898,0.022943018,0.0052801226,0.75155044],"study_design_scores_gemma":[0.000032010346,0.00008248608,0.000979432,0.000030153013,0.000031146745,0.00046057752,0.00015743398,0.9518961,0.005587383,0.03748089,0.0032409367,0.000021441721],"about_ca_topic_score_codex":0.0020089927,"about_ca_topic_score_gemma":0.0024419727,"teacher_disagreement_score":0.003555853,"about_ca_system_score_codex":0.0005963174,"about_ca_system_score_gemma":0.0013893018,"threshold_uncertainty_score":0.010915577},"labels":[],"label_agreement":null},{"id":"W2171332293","doi":"10.1145/1366102.1366103","title":"Conditional functional dependencies for capturing data inconsistencies","year":2008,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":458,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Bell (Canada)","funders":"Engineering and Physical Sciences Research Council","keywords":"Computer science; Functional dependency; Data integrity; Consistency (knowledge bases); Relational database; SQL; Data mining; Set (abstract data type); Schema (genetic algorithms); Database; Programming language; Theoretical computer science; Algorithm; Information retrieval; Artificial intelligence","score_opus":0.5917106512846138,"score_gpt":0.4156652633131301,"score_spread":0.17604538797148367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171332293","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014289715,0.0002522221,0.9796801,0.00044696988,0.0000645448,0.00018776473,0.0010892242,0.002497668,0.0014916924],"genre_scores_gemma":[0.24742,0.00033186268,0.7472166,0.00050925935,0.00009172968,0.00033564636,0.0022078482,0.00054350833,0.0013436428],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9896253,0.0024083557,0.0009249324,0.0018200872,0.0045621647,0.00065924093],"domain_scores_gemma":[0.9679843,0.01933556,0.0029470064,0.0053032762,0.0040651374,0.00036463712],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059604896,0.0014937815,0.001008972,0.003373366,0.0013902142,0.002483564,0.0036197691,0.0014716295,0.004050264],"category_scores_gemma":[0.032023925,0.0010279831,0.0019267814,0.0031944667,0.0020399545,0.0072788573,0.0031405154,0.0034816687,0.0005684242],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005351298,0.00030448032,0.020626577,0.001195576,0.00026014788,0.0013535963,0.0010277877,0.25480938,0.023103612,0.35385057,0.016543103,0.32639],"study_design_scores_gemma":[0.000062278945,0.00015816547,0.0023927263,0.00017919947,0.00014578446,0.00094339176,0.00025014495,0.75459087,0.033036456,0.18093623,0.027165236,0.00013950255],"about_ca_topic_score_codex":0.009169787,"about_ca_topic_score_gemma":0.011122894,"teacher_disagreement_score":0.009169787,"about_ca_system_score_codex":0.0021762538,"about_ca_system_score_gemma":0.003973315,"threshold_uncertainty_score":0.031522512},"labels":[],"label_agreement":null},{"id":"W2553567329","doi":"10.1145/3004295","title":"Smart Meter Data Analytics","year":2016,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Time Series Analysis and Forecasting","field":"Computer Science","cited_by":67,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Data analysis; Analytics; Smart meter; Big data; Data science; Data mining; Smart grid; Electrical engineering","score_opus":0.09289352135828077,"score_gpt":0.27501557937919624,"score_spread":0.18212205802091547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2553567329","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06512266,0.0034705854,0.6509171,0.0058530234,0.0015101524,0.0013376606,0.102508046,0.0765393,0.09274152],"genre_scores_gemma":[0.5547987,0.0039536282,0.27150142,0.0014277701,0.0006869314,0.00056033424,0.14333244,0.002841459,0.020897387],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975321,0.00032204684,0.00022834256,0.0005749078,0.0011775268,0.00016507096],"domain_scores_gemma":[0.99500436,0.00091576925,0.0003808374,0.0016273193,0.0019044887,0.00016719598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015045239,0.0015478653,0.0010885156,0.0028401506,0.0005625002,0.0029664258,0.0017775461,0.0007381661,0.010812725],"category_scores_gemma":[0.008826906,0.00042784266,0.00072550814,0.005655748,0.00036800577,0.004133721,0.0019369851,0.0012644381,0.009049088],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053414976,0.0004017917,0.02799456,0.00097323273,0.0002824021,0.00038397175,0.00033661057,0.071799435,0.013149627,0.039200142,0.257101,0.5878431],"study_design_scores_gemma":[0.00008415136,0.00018545963,0.015286529,0.00024673596,0.000096140626,0.0005062041,0.00057700416,0.5123786,0.042176984,0.07801474,0.35032296,0.00012447717],"about_ca_topic_score_codex":0.0035148624,"about_ca_topic_score_gemma":0.002756365,"teacher_disagreement_score":0.010812725,"about_ca_system_score_codex":0.00093874946,"about_ca_system_score_gemma":0.001276905,"threshold_uncertainty_score":0.03617221},"labels":[],"label_agreement":null},{"id":"W2740924709","doi":"10.1145/3068335","title":"DBSCAN Revisited, Revisited","year":2017,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":2625,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University; University of Alberta","funders":"","keywords":"Computer science; DBSCAN; Data mining; Heuristics; Information retrieval; Algorithm; Artificial intelligence; Cluster analysis","score_opus":0.03953632886572851,"score_gpt":0.28752780225235086,"score_spread":0.24799147338662236,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2740924709","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024258574,0.06059712,0.72930074,0.12361847,0.00963214,0.00017070812,0.0011939433,0.004991226,0.04623705],"genre_scores_gemma":[0.31812415,0.027146984,0.6015165,0.022838196,0.005715937,0.00021357174,0.0014516978,0.0016846528,0.021308294],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9756469,0.008321157,0.0012472882,0.0027973896,0.011138028,0.0008492405],"domain_scores_gemma":[0.96483886,0.01724369,0.0009839458,0.0059562232,0.009832587,0.0011446313],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02027269,0.0013793627,0.0027660355,0.0072846245,0.0027350034,0.014289488,0.0060949414,0.005047181,0.0055761165],"category_scores_gemma":[0.079933025,0.0013494715,0.0013766525,0.021336533,0.0072264904,0.0142529365,0.004312376,0.012469385,0.002477925],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058752944,0.0000773709,0.0018606429,0.0003581035,0.00019219858,0.0001801117,0.00046590125,0.032155957,0.00050224137,0.47007117,0.11061603,0.38293275],"study_design_scores_gemma":[0.00011497132,0.000120984485,0.00071800017,0.00032559194,0.00007723065,0.0009507195,0.0008897463,0.26025936,0.004560767,0.55482525,0.17701255,0.00014494923],"about_ca_topic_score_codex":0.023808187,"about_ca_topic_score_gemma":0.023367932,"teacher_disagreement_score":0.023808187,"about_ca_system_score_codex":0.0075089037,"about_ca_system_score_gemma":0.0070958,"threshold_uncertainty_score":0.1072135},"labels":[],"label_agreement":null},{"id":"W2766916843","doi":"10.1145/3110214","title":"Blazes","year":2017,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Distributed systems and fault tolerance","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Air Force Office of Scientific Research; Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Computer science; Debugging; Consistency (knowledge bases); Scalability; Distributed computing; Programming language; Database; Artificial intelligence","score_opus":0.03784811562260383,"score_gpt":0.288350591027029,"score_spread":0.25050247540442516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2766916843","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018104099,0.0019864016,0.39834058,0.0041402024,0.0017200253,0.00067939825,0.018431688,0.21996705,0.3366305],"genre_scores_gemma":[0.13888545,0.0023145822,0.26824355,0.0036659995,0.0005123036,0.00076159806,0.044067767,0.0471832,0.49436566],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998725,0.00015770226,0.00008425568,0.0002863727,0.00058527105,0.00016137911],"domain_scores_gemma":[0.9982772,0.00028461107,0.000107140346,0.0006670548,0.00053478597,0.00012930422],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012526929,0.0010970843,0.0005866899,0.001251886,0.0012119695,0.0032391194,0.002169505,0.0011789454,0.12657598],"category_scores_gemma":[0.0038946061,0.0007226714,0.0007934598,0.0008883757,0.0008143197,0.0048178555,0.0033950054,0.0020781315,0.07684606],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008645735,0.00019996821,0.0033592586,0.00087926706,0.00008109592,0.0003465731,0.00087289145,0.004466765,0.018140543,0.16761847,0.44459453,0.35857597],"study_design_scores_gemma":[0.00005149277,0.000053961954,0.00073983206,0.000095073585,0.000024174005,0.00023939842,0.00011656416,0.0075496137,0.009382621,0.021506313,0.9602034,0.00003758546],"about_ca_topic_score_codex":0.004654299,"about_ca_topic_score_gemma":0.0066144494,"teacher_disagreement_score":0.12657598,"about_ca_system_score_codex":0.0011936852,"about_ca_system_score_gemma":0.0018791505,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W3166524313","doi":"10.1145/3446980","title":"Optimizing One-time and Continuous Subgraph Queries using Worst-case Optimal Joins","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Graph Theory and Algorithms","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Joins; Query plan; Query optimization; Spatial query; Vertex (graph theory); Computation; Theoretical computer science; Graph; Sargable; Algorithm; Database; Web search query; Search engine; Information retrieval","score_opus":0.03081097010833498,"score_gpt":0.24837275789465005,"score_spread":0.21756178778631508,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3166524313","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.079781786,0.0006375266,0.90928215,0.0005463997,0.000064220854,0.00024570452,0.00046647782,0.0022308347,0.0067448583],"genre_scores_gemma":[0.4450021,0.0002994135,0.54986274,0.00019258112,0.000059953498,0.0002407262,0.0009073002,0.0007447496,0.0026903795],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9962053,0.00086631696,0.00018918808,0.00072882045,0.001415471,0.000594979],"domain_scores_gemma":[0.9970862,0.0017033081,0.00024226592,0.00047418437,0.00031312724,0.00018100892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025964652,0.0015618103,0.0015441969,0.0008175771,0.0007957364,0.002458572,0.0019460799,0.0010209044,0.0027350334],"category_scores_gemma":[0.0055252705,0.00063550856,0.0016029296,0.0015959806,0.0015643463,0.0035974958,0.0021619753,0.0018770599,0.0004317099],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035931254,0.0002692761,0.0017231804,0.00017814364,0.00009899172,0.00015542738,0.00021670105,0.8705576,0.0066692284,0.033021964,0.0035747613,0.08317532],"study_design_scores_gemma":[0.000036344827,0.000109692446,0.00019476847,0.000008450147,0.000024596573,0.000044639677,0.000084725936,0.973066,0.0027957424,0.022275575,0.0013466544,0.000012757833],"about_ca_topic_score_codex":0.008217824,"about_ca_topic_score_gemma":0.0099940235,"teacher_disagreement_score":0.008217824,"about_ca_system_score_codex":0.002204553,"about_ca_system_score_gemma":0.0027765092,"threshold_uncertainty_score":0.016339958},"labels":[],"label_agreement":null},{"id":"W3172623769","doi":"10.1145/3451159","title":"Graph Indexing for Efficient Evaluation of Label-constrained Reachability Queries","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Winnipeg","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reachability; Computer science; Vertex (graph theory); Search engine indexing; Transitive closure; Combinatorics; Graph; Theoretical computer science; Discrete mathematics; Mathematics; Information retrieval","score_opus":0.06869570056074788,"score_gpt":0.31231993113975526,"score_spread":0.2436242305790074,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3172623769","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.087747864,0.0017570045,0.8453371,0.00072369754,0.0001937649,0.00065179233,0.007715393,0.04781965,0.008053695],"genre_scores_gemma":[0.40250847,0.00050333457,0.5793366,0.00029210013,0.00011839129,0.00045174506,0.012925195,0.0016625843,0.0022016568],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99698883,0.00044592153,0.00034020565,0.0005408675,0.0013392831,0.0003449882],"domain_scores_gemma":[0.99347275,0.002859826,0.0004901552,0.0020871775,0.0008483431,0.00024172726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014288954,0.0012762643,0.0016634952,0.0031921414,0.0011049433,0.002818245,0.0029651327,0.0011315014,0.005496247],"category_scores_gemma":[0.010603084,0.00064741616,0.001119495,0.0055700727,0.00088560104,0.0077147945,0.0023321572,0.0014977056,0.0016579096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028150508,0.00073580653,0.006244778,0.0013918682,0.0002516806,0.00059817167,0.00081667246,0.088189244,0.08228969,0.060102653,0.054070648,0.7024937],"study_design_scores_gemma":[0.00023402795,0.000283384,0.0014444944,0.00006239052,0.0000863442,0.00050467567,0.00037068152,0.86473596,0.042480756,0.07646699,0.013249776,0.00008046638],"about_ca_topic_score_codex":0.0073620123,"about_ca_topic_score_gemma":0.010884592,"teacher_disagreement_score":0.0073620123,"about_ca_system_score_codex":0.002178643,"about_ca_system_score_gemma":0.0032189987,"threshold_uncertainty_score":0.018386722},"labels":[],"label_agreement":null},{"id":"W3212747350","doi":"10.1145/3483940","title":"On Directed Densest Subgraph Discovery","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Graph Neural Networks","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"University of Hong Kong; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Computer science; Scalability; Induced subgraph isomorphism problem; Enhanced Data Rates for GSM Evolution; Core (optical fiber); Graph; Directed graph; Subgraph isomorphism problem; Efficient algorithm; Theoretical computer science; Algorithm; Artificial intelligence; Database; Line graph","score_opus":0.019904467790991108,"score_gpt":0.25218275451067046,"score_spread":0.23227828671967934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3212747350","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033440087,0.0025198297,0.95139027,0.0020175944,0.00015991881,0.00041294433,0.0025699807,0.003544391,0.0039450293],"genre_scores_gemma":[0.19730648,0.0017132285,0.7843827,0.00083722745,0.0002764254,0.0003793303,0.009217323,0.0005950376,0.005292333],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964361,0.0008633066,0.000181354,0.0012271212,0.0009943385,0.00029774648],"domain_scores_gemma":[0.9907951,0.004780336,0.0008681868,0.0022019628,0.0010138465,0.00034054922],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002531888,0.0018173116,0.0021752382,0.0053232145,0.0017908016,0.0024547263,0.0033818346,0.0021393155,0.0032855086],"category_scores_gemma":[0.016998945,0.0013108373,0.0018062528,0.008397704,0.0019107759,0.0063310754,0.0036704824,0.002197367,0.0015134139],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043757702,0.00043497828,0.009328983,0.0011721901,0.0003538116,0.00060778414,0.0007827538,0.31724343,0.0050591594,0.0930625,0.056563962,0.5149528],"study_design_scores_gemma":[0.00007890674,0.00007170536,0.0012896002,0.000078656405,0.00007589885,0.0005020746,0.00029828036,0.80215275,0.0029857496,0.17678812,0.0156404,0.000037819704],"about_ca_topic_score_codex":0.01427237,"about_ca_topic_score_gemma":0.020549875,"teacher_disagreement_score":0.01427237,"about_ca_system_score_codex":0.002419546,"about_ca_system_score_gemma":0.003373355,"threshold_uncertainty_score":0.028378546},"labels":[],"label_agreement":null},{"id":"W4245823248","doi":"10.1145/1114244.1114250","title":"Optimization of query streams using semantic prefetching","year":2005,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Latency (audio); Context (archaeology); Server; Query optimization; STREAMS; Data mining; Information retrieval; Database; Computer network","score_opus":0.03100413104897517,"score_gpt":0.2743439780064526,"score_spread":0.24333984695747743,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4245823248","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5548804,0.00093433034,0.4227339,0.00065818895,0.00008899773,0.0002899914,0.0007340335,0.014973912,0.004706382],"genre_scores_gemma":[0.8958384,0.00017818606,0.10101369,0.00011394589,0.000048003887,0.00008303603,0.00076424266,0.00026276667,0.0016976296],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99903715,0.00017218437,0.00007690735,0.0002204392,0.00035805083,0.00013532679],"domain_scores_gemma":[0.99813026,0.0008649971,0.00014840189,0.00040845876,0.00036871008,0.00007916435],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00091900886,0.00084543234,0.00083599036,0.0006126515,0.00039553765,0.0009152305,0.0012000679,0.00049809675,0.0014795589],"category_scores_gemma":[0.005832576,0.00036942813,0.00033532106,0.0011272862,0.0004402454,0.0019122807,0.00085714564,0.000782013,0.00030321206],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001562968,0.00058296236,0.011739608,0.00016738976,0.0000954908,0.00030821678,0.00021761586,0.47619158,0.056792077,0.0045067235,0.0087213535,0.439114],"study_design_scores_gemma":[0.00003736059,0.0000623432,0.0006391784,0.0000019619301,0.00001092353,0.00003112136,0.000027877775,0.9846859,0.012297466,0.0016246208,0.00057314296,0.000008187427],"about_ca_topic_score_codex":0.005737296,"about_ca_topic_score_gemma":0.008995467,"teacher_disagreement_score":0.005737296,"about_ca_system_score_codex":0.00080168963,"about_ca_system_score_gemma":0.0014176213,"threshold_uncertainty_score":0.011407793},"labels":[],"label_agreement":null},{"id":"W4256119157","doi":"10.1145/1189769.1189778","title":"Peer data exchange","year":2006,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":69,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Data exchange; Tuple; Peer-to-peer; Schema (genetic algorithms); Data sharing; Generalization; Data source; Theoretical computer science; Information retrieval; Distributed computing; World Wide Web","score_opus":0.0697215860345619,"score_gpt":0.28923607189616385,"score_spread":0.21951448586160194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4256119157","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011753366,0.0030303325,0.8350553,0.0070961784,0.0016913593,0.0014003426,0.0030662478,0.00343111,0.13347587],"genre_scores_gemma":[0.29791114,0.005308218,0.5440571,0.004950394,0.0016051373,0.0019278886,0.014163565,0.0018734593,0.12820306],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9871514,0.0032764738,0.0014034563,0.0025881117,0.004486065,0.0010945421],"domain_scores_gemma":[0.98332536,0.0043334984,0.00094189466,0.008292156,0.0024256904,0.0006814611],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008101888,0.001024668,0.0014780761,0.0017032672,0.002705371,0.0074287597,0.006299261,0.0029650622,0.037993643],"category_scores_gemma":[0.025501918,0.0007195282,0.0017315098,0.0038623123,0.001912917,0.018629733,0.012749839,0.0032706978,0.011647761],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020055329,0.00010682851,0.0013101685,0.00049789937,0.00009301199,0.0004435896,0.0008353336,0.011144841,0.0010944186,0.78591985,0.055302612,0.14305095],"study_design_scores_gemma":[0.00009244546,0.00008378852,0.00029936168,0.0001549779,0.000046461897,0.0007383129,0.00048379696,0.027000275,0.0024853589,0.35530153,0.61325973,0.00005396669],"about_ca_topic_score_codex":0.0026075346,"about_ca_topic_score_gemma":0.0017368873,"teacher_disagreement_score":0.037993643,"about_ca_system_score_codex":0.0021432044,"about_ca_system_score_gemma":0.003359398,"threshold_uncertainty_score":0.12710136},"labels":[],"label_agreement":null},{"id":"W4309617049","doi":"10.1145/3571281","title":"Efficiently Cleaning Structured Event Logs: A Graph Repair Approach","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Beijing National Research Center For Information Science And Technology; National Natural Science Foundation of China","keywords":"Computer science; Workflow; Event (particle physics); Pruning; Data mining; Graph; Profiling (computer programming); Information retrieval; Theoretical computer science; Database; Programming language","score_opus":0.11679595709416458,"score_gpt":0.35897822892552084,"score_spread":0.24218227183135627,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4309617049","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012943939,0.00031629167,0.97890705,0.0004995849,0.00006408377,0.00028503343,0.0008466721,0.005584656,0.0005527845],"genre_scores_gemma":[0.10936054,0.00029419205,0.8830175,0.00024385795,0.0000641066,0.00021932139,0.0043263333,0.00066884054,0.0018052582],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99689853,0.0008022708,0.0002484814,0.0007986059,0.0010448595,0.00020732453],"domain_scores_gemma":[0.98578984,0.0066987714,0.0013627224,0.0039883745,0.0018717967,0.0002885988],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024219966,0.0018039519,0.0014773315,0.0037636007,0.0017017073,0.0014181735,0.0031884552,0.0016762712,0.001925949],"category_scores_gemma":[0.014458164,0.00086086453,0.0020587426,0.0033862463,0.0014028577,0.0035262823,0.0023803408,0.0022102916,0.0010711302],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004402264,0.00050727284,0.007456467,0.0011124125,0.00023219992,0.0010510787,0.0010649452,0.28516725,0.028561326,0.01839515,0.026071884,0.62993985],"study_design_scores_gemma":[0.000065524284,0.00015794206,0.0013293488,0.0000515255,0.000122218,0.0005202333,0.0004784882,0.9183147,0.015754899,0.0504079,0.012744082,0.000053291296],"about_ca_topic_score_codex":0.008777151,"about_ca_topic_score_gemma":0.013025486,"teacher_disagreement_score":0.008777151,"about_ca_system_score_codex":0.0010566204,"about_ca_system_score_gemma":0.0027838952,"threshold_uncertainty_score":0.01745212},"labels":[],"label_agreement":null},{"id":"W4392139269","doi":"10.1145/3649133","title":"Sharing Queries with Nonequivalent User-defined Aggregate Functions","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Rewriting; Computer science; Aggregate (composite); Class (philosophy); Query language; Function (biology); Materialized view; Extension (predicate logic); Theoretical computer science; Database query; Reuse; Programming language; Information retrieval; Artificial intelligence; View","score_opus":0.026689537956721236,"score_gpt":0.25969814733779084,"score_spread":0.2330086093810696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392139269","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09895613,0.00037984285,0.89235604,0.0003024645,0.000069215304,0.00015700005,0.00023068335,0.004731388,0.0028173118],"genre_scores_gemma":[0.69474125,0.00024679335,0.29968706,0.00030424935,0.000078305464,0.00013059158,0.0006433126,0.0012459776,0.0029224628],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98776054,0.0036362149,0.001595194,0.0016684488,0.004347802,0.0009918321],"domain_scores_gemma":[0.97336835,0.010247345,0.0013320915,0.012475062,0.002146706,0.0004304407],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009485825,0.00084518705,0.0015963567,0.00086781563,0.00097114727,0.0038174342,0.0031080141,0.0012296272,0.0015451426],"category_scores_gemma":[0.020337464,0.0007015517,0.0020416044,0.0011309303,0.0019133859,0.007458214,0.004144976,0.0020487742,0.00043514604],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018093642,0.0006966187,0.016060162,0.000853774,0.0007167416,0.0018968729,0.005172081,0.11247293,0.11157442,0.42761987,0.0069220657,0.31420502],"study_design_scores_gemma":[0.00012900963,0.0004955193,0.0019889006,0.00009090883,0.00043075328,0.0010568423,0.00083796796,0.57156587,0.18344888,0.19807321,0.04166315,0.00021904359],"about_ca_topic_score_codex":0.0022336196,"about_ca_topic_score_gemma":0.0019320355,"teacher_disagreement_score":0.009485825,"about_ca_system_score_codex":0.0015157777,"about_ca_system_score_gemma":0.0019947535,"threshold_uncertainty_score":0.050166428},"labels":[],"label_agreement":null},{"id":"W4410830018","doi":"10.1145/3736756","title":"Efficient Parallel Boolean Expression Matching","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Science and Technology Commission of Shanghai Municipality","keywords":"Computer science; Boolean expression; Expression (computer science); Matching (statistics); Regular expression; Theoretical computer science; Boolean function; Algorithm; Programming language; Mathematics","score_opus":0.01736739543533445,"score_gpt":0.26725246816992965,"score_spread":0.2498850727345952,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410830018","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04689043,0.00069160113,0.9331259,0.00045682804,0.000104310566,0.00030232494,0.0010915006,0.008818247,0.008518941],"genre_scores_gemma":[0.3570755,0.0005356852,0.6255549,0.0005622661,0.000084977895,0.00034041016,0.005157297,0.00085803034,0.009830938],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980222,0.0002004943,0.0002242284,0.00043027266,0.000861285,0.00026158665],"domain_scores_gemma":[0.99871016,0.00042135248,0.00010835912,0.00037168697,0.00034110094,0.00004741739],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008107847,0.0008546436,0.0010890181,0.0015686435,0.0007826638,0.0015531894,0.0018237247,0.0006562033,0.0053553274],"category_scores_gemma":[0.004309177,0.00031247607,0.0010380064,0.0037342352,0.00055199757,0.0038654506,0.0019581788,0.00078834966,0.0017980138],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044979152,0.00030251849,0.0028366682,0.0003580634,0.00007490815,0.00029714606,0.00028394925,0.033836182,0.052697234,0.03842133,0.02133904,0.8491033],"study_design_scores_gemma":[0.000136705,0.00022910051,0.00123843,0.000030339585,0.00009163177,0.0006357789,0.00025023503,0.7879449,0.10017872,0.0777259,0.031481866,0.000056483444],"about_ca_topic_score_codex":0.0029563732,"about_ca_topic_score_gemma":0.0035578671,"teacher_disagreement_score":0.0053553274,"about_ca_system_score_codex":0.001040266,"about_ca_system_score_gemma":0.0016738487,"threshold_uncertainty_score":0.017915368},"labels":[],"label_agreement":null},{"id":"W4415991463","doi":"10.1145/3774322","title":"Proof-of-Execution: Low-Latency Consensus via Speculative Execution","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Distributed systems and fault tolerance","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Protocol (science); Latency (audio); Key (lock); Distributed transaction; Transaction processing; Two-phase commit protocol; Database transaction","score_opus":0.01594775058131951,"score_gpt":0.2646067206446487,"score_spread":0.24865897006332918,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415991463","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023813194,0.000277902,0.97009695,0.00049421855,0.00012112779,0.00013952111,0.000038789087,0.001147195,0.0038710877],"genre_scores_gemma":[0.78869605,0.00045979774,0.20493527,0.00022856283,0.0001106347,0.00031966504,0.00015200322,0.00030531376,0.0047926577],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99631566,0.0013713972,0.00022820572,0.0005237276,0.0011222123,0.00043876696],"domain_scores_gemma":[0.9874315,0.0060184,0.00110172,0.003744762,0.0011536704,0.0005498887],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043888576,0.0007286519,0.0008236761,0.0007298107,0.0011604875,0.0018221508,0.0026780344,0.0011625479,0.0030936836],"category_scores_gemma":[0.015666796,0.0006180533,0.0006072868,0.00075118145,0.002707676,0.0044044075,0.004804255,0.0026274272,0.0008341861],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009623872,0.00024409496,0.0017871549,0.00045998322,0.00012020397,0.0011128123,0.0011892956,0.27717283,0.055332467,0.5345024,0.0056712986,0.12144503],"study_design_scores_gemma":[0.000109400935,0.0001489599,0.00011604051,0.00003529103,0.000026851894,0.00014136432,0.00008567076,0.81865597,0.019228578,0.15514548,0.0062664924,0.00003994763],"about_ca_topic_score_codex":0.0012206802,"about_ca_topic_score_gemma":0.0011177362,"teacher_disagreement_score":0.0043888576,"about_ca_system_score_codex":0.0008743202,"about_ca_system_score_gemma":0.0021717884,"threshold_uncertainty_score":0.023210764},"labels":[],"label_agreement":null},{"id":"W4417496854","doi":"10.1145/3785662","title":"Semi-Oblivious Chase Termination for Linear Existential Rules and Beyond: An Experimental Analysis","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Existentialism; Chase; Set (abstract data type); Generalization; Linearization; Class (philosophy); Focus (optics)","score_opus":0.023561886560447394,"score_gpt":0.30921964395633783,"score_spread":0.28565775739589044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417496854","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7630474,0.003720193,0.17480701,0.0019415416,0.0005833557,0.0011677969,0.0028749479,0.0076343957,0.044223417],"genre_scores_gemma":[0.8824702,0.0008589999,0.10427738,0.00050121074,0.0001377792,0.0007728923,0.0034994523,0.0013819742,0.0061000227],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99028695,0.002798481,0.00069724466,0.0015013992,0.0034896361,0.0012262996],"domain_scores_gemma":[0.9106302,0.06457847,0.0026589553,0.015800431,0.0050548604,0.0012770223],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007777838,0.0010739425,0.0014745754,0.000977125,0.0015424136,0.0024966886,0.0030834663,0.0020103345,0.012126594],"category_scores_gemma":[0.05304022,0.0005569426,0.0011919258,0.0014902991,0.0022667074,0.0059846262,0.0026697982,0.0047636745,0.0025403048],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.010777949,0.012893413,0.017913822,0.004457474,0.00078371423,0.0014008033,0.002700428,0.27429795,0.09989347,0.20839877,0.05543042,0.31105182],"study_design_scores_gemma":[0.0007167508,0.0019419752,0.0023630064,0.00019447766,0.00016341277,0.00045326384,0.0006705613,0.8380319,0.053997185,0.0893058,0.012053585,0.000108084474],"about_ca_topic_score_codex":0.0020856436,"about_ca_topic_score_gemma":0.0022089395,"teacher_disagreement_score":0.012126594,"about_ca_system_score_codex":0.0020013133,"about_ca_system_score_gemma":0.0026971276,"threshold_uncertainty_score":0.041133642},"labels":[],"label_agreement":null},{"id":"W7117687674","doi":"10.1145/3786785","title":"Finding Smallest Witnesses for Conjunctive Queries","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Conjunctive query; Rewriting; Property (philosophy); Logarithm; Class (philosophy); Time complexity; Polynomial; Constant (computer programming)","score_opus":0.03221601730685047,"score_gpt":0.2997566851307163,"score_spread":0.2675406678238658,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117687674","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14949176,0.0013273705,0.831946,0.003451832,0.00016503505,0.0005278337,0.0032598241,0.0046539293,0.0051764078],"genre_scores_gemma":[0.5633989,0.00084292056,0.42470145,0.0008864978,0.00020757102,0.0003641118,0.0052617057,0.0009452009,0.0033915383],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99349135,0.00095906446,0.00067256,0.002473644,0.0016640885,0.00073924754],"domain_scores_gemma":[0.96483773,0.02394155,0.0017078595,0.0071653477,0.0015717903,0.0007756352],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004470903,0.0015855831,0.0020711855,0.0010066187,0.0015611709,0.005324111,0.0038870473,0.002000483,0.007540354],"category_scores_gemma":[0.03845896,0.0013682396,0.002752077,0.0021589776,0.0022270032,0.018925123,0.004916113,0.0041498817,0.0012536395],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031803239,0.0008722564,0.01611427,0.0038825856,0.0007940118,0.0013272882,0.0045129675,0.1840881,0.046349373,0.34435418,0.0237843,0.37074032],"study_design_scores_gemma":[0.0003026444,0.00027633534,0.0010954109,0.00016886451,0.00030233513,0.0008534955,0.001087873,0.47465667,0.025843604,0.48551613,0.0098130815,0.00008360172],"about_ca_topic_score_codex":0.0024345268,"about_ca_topic_score_gemma":0.0031557204,"teacher_disagreement_score":0.007540354,"about_ca_system_score_codex":0.002093408,"about_ca_system_score_gemma":0.002685674,"threshold_uncertainty_score":0.025224984},"labels":[],"label_agreement":null}]}