{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":1336,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":1336,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"3b3b69ad4986","filters":{"topic":"Algorithms and Data Compression"}},"results":[{"id":"W3099878876","doi":"","title":"Array programming with NumPy","year":2020,"lang":"en","type":"review","venue":"TUScholarShare (Temple University)","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":18805,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia; University of Toronto","funders":"","keywords":"Computer science","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.05573241774715602,"gpt":0.2719552118369395,"spread":0.2162227940897835,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00231303,0.001574665,0.001966017,0.002358736,0.0007822433,0.003201129,0.003368858,0.001257622,0.04514243],"category_scores_gemma":[0.008432261,0.0008436229,0.001924546,0.004449083,0.001092922,0.003812614,0.002742066,0.004174584,0.03897416],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006842637,"about_ca_system_score_gemma":0.00223677,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001043527,"about_ca_topic_score_gemma":0.0008176706,"domain_scores_codex":[0.9980935,0.0005325556,0.000225199,0.0002740727,0.0007477188,0.0001268818],"domain_scores_gemma":[0.9973004,0.001308979,0.0001945721,0.0003761142,0.0006369083,0.0001829794],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001560582,0.00005586939,0.0004199962,0.006175043,0.0001796587,0.0002386351,0.0003064382,0.006310891,0.003354548,0.1075193,0.352216,0.5230676],"study_design_scores_gemma":[0.00003381457,0.00001764253,0.0001407715,0.0004673984,0.00002948836,0.0002149767,0.0000219525,0.004136188,0.002491252,0.0329811,0.959426,0.00003944188],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.001554871,0.07629847,0.7759426,0.005622879,0.002790075,0.0004267168,0.008101626,0.05540373,0.07385895],"genre_scores_gemma":[0.03054616,0.1441496,0.7285653,0.00707786,0.002528384,0.003322498,0.01706895,0.02739236,0.03934889],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.04514243,"threshold_uncertainty_score":0.1510165,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2086959852","doi":"10.1007/s10994-009-5103-0","title":"NP-hardness of Euclidean sum-of-squares clustering","year":2009,"lang":"en","type":"article","venue":"Machine Learning","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":863,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Group for Research in Decision Analysis; HEC Montréal; Polytechnique Montréal","funders":"","keywords":"Cluster analysis; Mathematics; Euclidean distance; Explained sum of squares; Euclidean geometry; Combinatorics; Pattern recognition (psychology); Artificial intelligence; Computer science; Statistics","authors":[{"name":"Daniel Aloise","is_ca":true},{"name":"Amit Deshpande","is_ca":false},{"name":"Pierre Hansen","is_ca":true},{"name":"Preyas Popat","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01257153537582592,"gpt":0.2570000966085654,"spread":0.2444285612327395,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002537121,0.001411517,0.003681193,0.001009609,0.002140614,0.004627993,0.005001582,0.003719729,0.007659008],"category_scores_gemma":[0.02615032,0.001245204,0.0015418,0.003963928,0.003144532,0.008856873,0.003722825,0.005054315,0.001836969],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003060328,"about_ca_system_score_gemma":0.003614254,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005403852,"about_ca_topic_score_gemma":0.005508895,"domain_scores_codex":[0.9941329,0.001996443,0.0003830916,0.001290274,0.001659982,0.0005372186],"domain_scores_gemma":[0.9765046,0.01788844,0.000920865,0.002771514,0.001392822,0.0005217586],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001748311,0.0004490148,0.002109059,0.001522197,0.0003359361,0.0003847281,0.0008351782,0.5310929,0.003592115,0.171881,0.08371957,0.2023299],"study_design_scores_gemma":[0.0002152641,0.00005834208,0.0004743753,0.00005793379,0.00004826103,0.0002916596,0.0002854683,0.526195,0.001816831,0.4641044,0.006420189,0.00003229407],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09082035,0.005213303,0.8392804,0.01782815,0.0007786136,0.0003265899,0.004386668,0.002989387,0.03837655],"genre_scores_gemma":[0.6302361,0.002875429,0.3351007,0.002868867,0.0009604881,0.0007316512,0.005392831,0.001430299,0.02040365],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007659008,"threshold_uncertainty_score":0.02562195,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2167155938","doi":"10.1101/gr.094607.109","title":"JBrowse: A next-generation genome browser","year":2009,"lang":"en","type":"article","venue":"Genome Research","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":776,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Ontario Institute for Cancer Research","funders":"National Human Genome Research Institute; National Institutes of Health","keywords":"JavaScript; Upload; Computer science; Web server; Genome; Genome browser; Annotation; Zoom; Panning (audio); Biology; Rendering (computer graphics); Web browser; World Wide Web; The Internet; Genomics; Genetics; Artificial intelligence","authors":[{"name":"Mitchell E. Skinner","is_ca":false},{"name":"Andrew Uzilov","is_ca":false},{"name":"Lincoln Stein","is_ca":true},{"name":"Chris Mungall","is_ca":false},{"name":"Ian Holmes","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2085239895237908,"gpt":0.3744306107962284,"spread":0.1659066212724375,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001510609,0.0008037914,0.0006523883,0.001449629,0.0005907391,0.002097674,0.002461988,0.00120644,0.01744792],"category_scores_gemma":[0.003560039,0.0009719798,0.0008006937,0.001250328,0.0003123152,0.0025189,0.001512891,0.002285276,0.01403789],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004862517,"about_ca_system_score_gemma":0.001337619,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004970572,"about_ca_topic_score_gemma":0.008918373,"domain_scores_codex":[0.9992961,0.00007395936,0.00005077677,0.0001190432,0.0003971044,0.00006301326],"domain_scores_gemma":[0.9988624,0.0003264497,0.00006283149,0.0002548661,0.0003355402,0.0001579008],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005904399,0.0002641205,0.002683003,0.0009042242,0.0001479842,0.0005636583,0.0005291175,0.005866037,0.07081655,0.02374547,0.4246746,0.4692148],"study_design_scores_gemma":[0.0001188594,0.00005261862,0.001374131,0.0001250702,0.00004413291,0.001006391,0.00009505493,0.03433325,0.04498181,0.0123329,0.9053959,0.0001399525],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.005783373,0.001893238,0.7451709,0.0007279731,0.0004948317,0.0001730229,0.01368783,0.2193892,0.01267955],"genre_scores_gemma":[0.02629942,0.002509624,0.8671907,0.0007176566,0.0001214118,0.0003955478,0.05297352,0.02644542,0.02334657],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.01744792,"threshold_uncertainty_score":0.0583691,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1997102766","doi":"10.1145/1882471.1882478","title":"A brief survey on sequence classification","year":2010,"lang":"en","type":"article","venue":"ACM SIGKDD Explorations Newsletter","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":552,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Sequence (biology); Feature selection; Artificial intelligence; Feature vector; Task (project management); Feature (linguistics); Machine learning; Pattern recognition (psychology); One-class classification; Support vector machine; Data mining","authors":[{"name":"Zhengzheng Xing","is_ca":true},{"name":"Jian Pei","is_ca":true},{"name":"Eamonn Keogh","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1153428195483115,"gpt":0.3164656778829131,"spread":0.2011228583346016,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002127419,0.001182812,0.002000881,0.007935201,0.0009974279,0.002917623,0.001995204,0.001630756,0.007648352],"category_scores_gemma":[0.007987824,0.0004630765,0.001290355,0.01522422,0.0006779229,0.005571523,0.001210836,0.001669289,0.007023884],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001071644,"about_ca_system_score_gemma":0.001988125,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003623365,"about_ca_topic_score_gemma":0.002026396,"domain_scores_codex":[0.9973118,0.0005059847,0.0003069131,0.0004640861,0.001261459,0.0001497886],"domain_scores_gemma":[0.9958704,0.001991598,0.0002209253,0.0004183938,0.001366528,0.000132053],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00006768613,0.00006738979,0.001249523,0.001072276,0.00004331065,0.00006980303,0.00006038824,0.003780375,0.0008701941,0.009171241,0.02954428,0.9540035],"study_design_scores_gemma":[0.00004122074,0.0003405764,0.004881024,0.001764426,0.0001214381,0.001963628,0.00031133,0.1099377,0.005453723,0.09812224,0.7769269,0.0001358127],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.007639334,0.4490373,0.4988672,0.004766244,0.004532092,0.0004299197,0.002452848,0.002400001,0.02987514],"genre_scores_gemma":[0.0650301,0.5727649,0.310064,0.002997578,0.01037024,0.0008407781,0.0122806,0.0005897784,0.02506206],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.007935201,"threshold_uncertainty_score":0.02558625,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1974033543","doi":"10.1145/1290672.1290680","title":"Succinct indexable dictionaries with applications to encoding <i>k</i> -ary trees, prefix sums and multisets","year":2007,"lang":"en","type":"article","venue":"ACM Transactions on Algorithms","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":378,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Prefix; Encoding (memory); Trie; Computer science; Prefix code; Theoretical computer science; Tree (set theory); Combinatorics; Mathematics; Data structure; Algorithm; Decoding methods; Artificial intelligence; Programming language","authors":[{"name":"Rajeev Raman","is_ca":false},{"name":"Venkatesh Raman","is_ca":false},{"name":"Srinivasa Rao Satti","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0167703244268879,"gpt":0.2608292871613097,"spread":0.2440589627344218,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006384462,0.0004288735,0.0008815259,0.000743105,0.0006444185,0.001819245,0.001360328,0.0009118584,0.004591396],"category_scores_gemma":[0.004151108,0.0004217413,0.0005690484,0.002719855,0.001234524,0.005531064,0.002223447,0.001960989,0.001260577],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001253064,"about_ca_system_score_gemma":0.00075938,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001221493,"about_ca_topic_score_gemma":0.002226557,"domain_scores_codex":[0.9992946,0.0001186533,0.0001119366,0.0001353348,0.0002537402,0.00008565549],"domain_scores_gemma":[0.9976215,0.0008260115,0.0002053002,0.0009633505,0.0002830609,0.0001008675],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0008690734,0.0003108799,0.001995478,0.0005548705,0.00004369181,0.0005333515,0.001347199,0.1176647,0.02889325,0.4946411,0.01588401,0.3372625],"study_design_scores_gemma":[0.000128855,0.000336973,0.000580102,0.0001845365,0.00004968831,0.0007025794,0.0005576719,0.4574439,0.04723132,0.4476316,0.04504282,0.0001100177],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1409972,0.00146971,0.8388299,0.002053522,0.0002689703,0.0002094847,0.001949485,0.003230056,0.01099163],"genre_scores_gemma":[0.386604,0.0009689777,0.6023071,0.0004864769,0.0001478446,0.0002539277,0.00226941,0.0003606101,0.0066016],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004591396,"threshold_uncertainty_score":0.01535976,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2142857718","doi":"10.1287/opre.50.6.1073.358","title":"An Object-Oriented Random-Number Package with Many Long Streams and Substreams","year":2002,"lang":"en","type":"article","venue":"Operations Research","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":343,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Disjoint sets; Generator (circuit theory); Random number generation; Reduction (mathematics); Implementation; Variance reduction; Sequence (biology); Set (abstract data type); Java; Variance (accounting); Algorithm; Parallel computing; Mathematics; Discrete mathematics; Programming language; Statistics; Power (physics)","authors":[{"name":"Pierre L’Ecuyer","is_ca":true},{"name":"Richard Simard","is_ca":true},{"name":"E. Jack Chen","is_ca":false},{"name":"W. David Kelton","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03584947236627194,"gpt":0.3315742707934994,"spread":0.2957247984272275,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004502065,0.001864601,0.001800455,0.002115965,0.0007020931,0.002187812,0.003640621,0.001614786,0.08680658],"category_scores_gemma":[0.01483436,0.001629837,0.001532034,0.0020741,0.0006272362,0.002554134,0.002167364,0.002640482,0.04573571],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006206817,"about_ca_system_score_gemma":0.001453367,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009585771,"about_ca_topic_score_gemma":0.001343077,"domain_scores_codex":[0.9982162,0.0004178273,0.000236557,0.0002192205,0.0007573993,0.0001527793],"domain_scores_gemma":[0.9940683,0.002897826,0.0004288789,0.001235253,0.00115299,0.0002168107],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007789698,0.0004278698,0.003265784,0.001534104,0.0003040559,0.0007871981,0.0003304045,0.04739293,0.0106317,0.09541203,0.4232705,0.4158645],"study_design_scores_gemma":[0.00101573,0.0002463672,0.001628641,0.0002822707,0.0001642553,0.001079924,0.0000320801,0.2963406,0.01814852,0.0903409,0.5904532,0.0002675675],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0006121069,0.00009749547,0.8818346,0.0001151764,0.0001622895,0.0003137705,0.003357328,0.1095064,0.004000775],"genre_scores_gemma":[0.01765116,0.0005104912,0.88263,0.000483993,0.0002793502,0.003505213,0.01232571,0.0621554,0.02045873],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.08680658,"threshold_uncertainty_score":0.2903969,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1976682045","doi":"10.1145/1242471.1242472","title":"A taxonomy of suffix array construction algorithms","year":2014,"lang":"en","type":"review","venue":"Minerva Access (University of Melbourne)","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":307,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Suffix array; Suffix; Algorithm; Implementation; Compressed suffix array; Suffix tree; Generalized suffix tree; Data structure; Theoretical computer science; Programming language","authors":[{"name":"W.F. Smyth","is_ca":true},{"name":"Andrew Turpin","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07356066266408752,"gpt":0.287019720484159,"spread":0.2134590578200715,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001666521,0.001439439,0.00142759,0.006690246,0.0008310988,0.002639622,0.002715476,0.001741281,0.004892101],"category_scores_gemma":[0.008266757,0.0008822663,0.001037949,0.01429329,0.0009972426,0.005633964,0.001425033,0.002188513,0.007496295],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009309998,"about_ca_system_score_gemma":0.002206843,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009388595,"about_ca_topic_score_gemma":0.0008257874,"domain_scores_codex":[0.9983053,0.0002583206,0.0002479324,0.0003716235,0.0007042147,0.0001126912],"domain_scores_gemma":[0.9956452,0.002424947,0.0002491011,0.0004253284,0.001170704,0.00008472216],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004365411,0.00006816702,0.0004570475,0.003545283,0.00003211708,0.00006310386,0.0001126116,0.002118079,0.002094531,0.02679836,0.01705674,0.9476104],"study_design_scores_gemma":[0.00003831621,0.0002084703,0.001031877,0.002489553,0.00008392055,0.002876383,0.0002425435,0.01551652,0.01092396,0.05630221,0.9101684,0.0001179203],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.005827402,0.5717989,0.37517,0.002407015,0.001175399,0.0005568641,0.001269096,0.002738993,0.03905632],"genre_scores_gemma":[0.02054299,0.5449899,0.4151386,0.001524263,0.001078128,0.0007496293,0.004045807,0.0005585673,0.01137218],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.006690246,"threshold_uncertainty_score":0.01636577,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2047386086","doi":"10.1145/506147.506150","title":"On the closest string and substring problems","year":2002,"lang":"en","type":"article","venue":"Journal of the ACM","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":237,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Western University; University of Waterloo","funders":"","keywords":"Substring; Hamming distance; Combinatorics; String (physics); Approximate string matching; Mathematics; String searching algorithm; String metric; Edit distance; Discrete mathematics; Set (abstract data type); Computer science; Algorithm; Pattern matching; Artificial intelligence","authors":[{"name":"Ming Li","is_ca":true},{"name":"Bin Ma","is_ca":true},{"name":"Lusheng Wang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03950130813966558,"gpt":0.2181354447543569,"spread":0.1786341366146914,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004705356,0.002735355,0.003858953,0.003485045,0.003116192,0.004420795,0.005137142,0.005322376,0.01227862],"category_scores_gemma":[0.03338671,0.001095331,0.002508814,0.01004381,0.005448339,0.01993572,0.006539696,0.008332925,0.004292056],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002790643,"about_ca_system_score_gemma":0.002128113,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003003068,"about_ca_topic_score_gemma":0.001770585,"domain_scores_codex":[0.9931073,0.002262999,0.0004722649,0.001631538,0.001937905,0.0005881224],"domain_scores_gemma":[0.9790701,0.01627814,0.0008306561,0.002288523,0.001003204,0.0005293948],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000790893,0.0003868366,0.001309627,0.000806519,0.0001553201,0.0003299323,0.000598982,0.1660438,0.001250052,0.544903,0.04592346,0.2375017],"study_design_scores_gemma":[0.000105861,0.00009279431,0.0001786048,0.0001049028,0.00003601711,0.0002532779,0.0001429365,0.1531081,0.0009227722,0.8295049,0.01551029,0.00003963572],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02192403,0.01149669,0.9340395,0.008127998,0.0009029029,0.0002402139,0.0008362872,0.001216461,0.0212158],"genre_scores_gemma":[0.2131836,0.01410462,0.7360367,0.003697814,0.003475992,0.0007446285,0.00411024,0.001135226,0.02351131],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01227862,"threshold_uncertainty_score":0.04107606,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1964541472","doi":"10.1016/s0890-5401(03)00057-9","title":"Distinguishing string selection problems","year":2003,"lang":"en","type":"article","venue":"Information and Computation","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":236,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Western University; University of Waterloo","funders":"","keywords":"Substring; String (physics); String metric; String searching algorithm; Approximate string matching; Mathematics; Combinatorics; Commentz-Walter algorithm; Time complexity; Edit distance; Set (abstract data type); Hamming distance; Discrete mathematics; Algorithm; Computer science; Pattern matching; Artificial intelligence","authors":[{"name":"J. Kevin Lanctot","is_ca":true},{"name":"Ming Li","is_ca":true},{"name":"Bin Ma","is_ca":true},{"name":"Shaojiu Wang","is_ca":false},{"name":"Louxin Zhang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0130856715713156,"gpt":0.2360714225352947,"spread":0.2229857509639791,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004320655,0.0008167509,0.001471128,0.002288539,0.001730556,0.004681414,0.002385586,0.004270988,0.01228019],"category_scores_gemma":[0.02623585,0.0006582581,0.001543945,0.003341868,0.003139411,0.01230268,0.00541569,0.00613305,0.002485572],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001177171,"about_ca_system_score_gemma":0.00101124,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002012132,"about_ca_topic_score_gemma":0.0001865615,"domain_scores_codex":[0.994379,0.001954701,0.0003602635,0.001316522,0.001506008,0.0004834622],"domain_scores_gemma":[0.9665434,0.02685403,0.0009372442,0.003854993,0.001064735,0.0007456481],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000423408,0.0001603554,0.001491378,0.0002354122,0.00004148676,0.0001899447,0.0002341999,0.006250408,0.001652357,0.8672848,0.0136678,0.1083686],"study_design_scores_gemma":[0.00002953851,0.00003701064,0.000261611,0.00002374396,0.00001597838,0.0001736391,0.00005786261,0.02800006,0.001797052,0.9642855,0.005306372,0.00001158387],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1512481,0.003092619,0.7624175,0.01383813,0.0009079062,0.0002078106,0.0009716353,0.001227905,0.06608844],"genre_scores_gemma":[0.8170556,0.001784148,0.1430096,0.002267529,0.001955765,0.0002681713,0.00339277,0.0005519689,0.02971443],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01228019,"threshold_uncertainty_score":0.04108137,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1556137908","doi":"","title":"Automatic algorithm configuration based on local search","year":2007,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":235,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Heuristics; Computer science; Algorithm; Local search (optimization); Heuristic; Search algorithm; Mathematical optimization; Flexibility (engineering); Mathematics; Artificial intelligence","authors":[{"name":"Frank Hutter","is_ca":true},{"name":"Holger H. Hoos","is_ca":true},{"name":"Thomas Stützle","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01561185102377989,"gpt":0.273821400177054,"spread":0.2582095491532741,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00523827,0.001903272,0.002253164,0.002709056,0.001430856,0.002762428,0.003813121,0.002168662,0.005760484],"category_scores_gemma":[0.03058321,0.001200411,0.0008793691,0.001790238,0.002044545,0.00374894,0.003275585,0.002304123,0.003049673],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001166736,"about_ca_system_score_gemma":0.001540782,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000822422,"about_ca_topic_score_gemma":0.001217899,"domain_scores_codex":[0.9921157,0.00348341,0.0006740905,0.001541967,0.001699995,0.0004847381],"domain_scores_gemma":[0.9848069,0.00778637,0.001042073,0.004240686,0.001860492,0.0002634228],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001265504,0.0003687584,0.00503841,0.0007628044,0.0002563535,0.0004759081,0.0008322999,0.3614453,0.03552613,0.028487,0.01051497,0.5550265],"study_design_scores_gemma":[0.0002512655,0.0002400947,0.0007520751,0.0001134052,0.00008821393,0.0004484383,0.0001301872,0.9367786,0.03207092,0.02266973,0.006353297,0.0001038528],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.026638,0.0003656287,0.9571081,0.0001100747,0.00004541372,0.0002631757,0.00007344226,0.01061409,0.004782073],"genre_scores_gemma":[0.3728013,0.0001681559,0.6217779,0.0001862982,0.00003593344,0.0007613899,0.0003396874,0.002659369,0.001269848],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005760484,"threshold_uncertainty_score":0.02770293,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1791987072","doi":"10.1002/spe.2203","title":"Decoding billions of integers per second through vectorization","year":2013,"lang":"en","type":"article","venue":"Software Practice and Experience","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":227,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Université TÉLUQ","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Decoding methods; Vectorization (mathematics); Scheme (mathematics); Encoding (memory); Data compression; Compression (physics)","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.01574918514338506,"gpt":0.2803758415423402,"spread":0.2646266563989551,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006007591,0.001046308,0.0004706507,0.001334714,0.0003849333,0.001129654,0.0009398051,0.0004814113,0.006816084],"category_scores_gemma":[0.002896723,0.0003201948,0.0003306268,0.001914738,0.0005864769,0.001870518,0.001360931,0.0007055185,0.003715346],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004706003,"about_ca_system_score_gemma":0.0008847202,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001403181,"about_ca_topic_score_gemma":0.001511983,"domain_scores_codex":[0.9992919,0.0000953589,0.00008126787,0.0000946104,0.0003772395,0.0000596152],"domain_scores_gemma":[0.9991148,0.0002381349,0.00007586155,0.0002796312,0.0002696565,0.00002194504],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0005567013,0.00007000194,0.0007115474,0.000225372,0.00003371997,0.0001437626,0.0002705276,0.01228167,0.06764778,0.02612046,0.01291351,0.8790249],"study_design_scores_gemma":[0.0002391166,0.0006060284,0.001280927,0.0001703044,0.00006834355,0.001041934,0.0003386968,0.4025892,0.369895,0.05301471,0.1706289,0.0001268719],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04711363,0.001695903,0.9232878,0.0005004415,0.000362676,0.0002379759,0.0005461726,0.01506042,0.01119499],"genre_scores_gemma":[0.2423444,0.001474154,0.7368845,0.000305047,0.0001449936,0.0003629501,0.002423597,0.001079748,0.01498066],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006816084,"threshold_uncertainty_score":0.02280211,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2147492358","doi":"10.1093/bioinformatics/18.12.1696","title":"DNACompress: fast and effective DNA sequence compression","year":2002,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":210,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Bioinformatics Solutions (Canada); Western University","funders":"National Science Foundation","keywords":"Compression (physics); Sequence (biology); DNA; Computer science; DNA sequencing; Data compression; Computational biology; Algorithm; Genetics; Biology; Materials science; Composite material","authors":[{"name":"Xin Chen","is_ca":false},{"name":"Ming Li","is_ca":false},{"name":"Bin Ma","is_ca":true},{"name":"John Tromp","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02148647027400059,"gpt":0.238584043247377,"spread":0.2170975729733765,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007324831,0.001168656,0.000558182,0.002020124,0.0005509713,0.0008478376,0.00125131,0.0007845041,0.0144094],"category_scores_gemma":[0.002705283,0.0005453086,0.0004075319,0.001817557,0.000492989,0.00103201,0.00120727,0.001170706,0.007305346],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000450183,"about_ca_system_score_gemma":0.0007034313,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009566056,"about_ca_topic_score_gemma":0.001350841,"domain_scores_codex":[0.9994324,0.00007205144,0.00004557052,0.0001039102,0.0003052807,0.00004085179],"domain_scores_gemma":[0.9991201,0.0003545757,0.00008737313,0.000165911,0.000213797,0.00005820177],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007889232,0.0001256345,0.0008934613,0.0005627236,0.00007034222,0.0004213182,0.0001618163,0.01275641,0.08729523,0.01486977,0.1096655,0.7723889],"study_design_scores_gemma":[0.0005099918,0.0003192596,0.001916713,0.0001508771,0.00007012073,0.00132957,0.00009356661,0.3799454,0.471835,0.02325881,0.1204643,0.0001064242],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03246675,0.002087807,0.8701677,0.0007798751,0.0004759717,0.0003362939,0.005522962,0.07769547,0.01046714],"genre_scores_gemma":[0.1188221,0.0009165324,0.8488306,0.0003600698,0.000240984,0.0007202437,0.01241429,0.003737742,0.01395739],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0144094,"threshold_uncertainty_score":0.04820424,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2119878143","doi":"10.1109/dcc.1997.582019","title":"A corpus for the evaluation of lossless compression algorithms","year":2002,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":201,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Lossless compression; Computer science; Data compression; Compression (physics); Algorithm; Lossy compression; Natural language processing; Compression ratio; The Internet; Artificial intelligence; Information retrieval; World Wide Web","authors":[{"name":"Ross Arnold","is_ca":false},{"name":"Tim Bell","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.125841513504027,"gpt":0.326270749231966,"spread":0.200429235727939,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005249115,0.001339544,0.001351477,0.008991401,0.003543436,0.002151984,0.002712629,0.001785261,0.01627619],"category_scores_gemma":[0.04688392,0.0006589828,0.0007654684,0.01091591,0.002545881,0.003041453,0.002982469,0.002032705,0.006437319],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002598722,"about_ca_system_score_gemma":0.002398358,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01160944,"about_ca_topic_score_gemma":0.01674316,"domain_scores_codex":[0.9894124,0.003201853,0.001412732,0.001002439,0.004680838,0.0002897019],"domain_scores_gemma":[0.9313816,0.03340641,0.001935561,0.01162273,0.02038937,0.001264368],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009882909,0.001488114,0.004758611,0.006948742,0.0001688317,0.0015021,0.004133499,0.01106961,0.02394697,0.01644052,0.427819,0.5007358],"study_design_scores_gemma":[0.0009492924,0.001264109,0.05441325,0.001614226,0.0002021327,0.004539313,0.00401463,0.04522739,0.06345784,0.01339212,0.8104768,0.0004489063],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.362641,0.01556206,0.2115877,0.004254941,0.002684238,0.01323947,0.2718037,0.01637288,0.101854],"genre_scores_gemma":[0.2563574,0.00392528,0.2779827,0.0007410576,0.000758428,0.01370693,0.4103568,0.003941732,0.03222965],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01627619,"threshold_uncertainty_score":0.05444926,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1986633843","doi":"10.1006/jcss.2002.1822","title":"Optimal Bounds for the Predecessor Problem and Related Problems","year":2002,"lang":"en","type":"article","venue":"Journal of Computer and System Sciences","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":191,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Matching (statistics); Set (abstract data type); Computer science; Element (criminal law); Class (philosophy); Dynamic problem; Multiplication (music); Upper and lower bounds; Mathematics; Algorithm; Mathematical optimization; Combinatorics","authors":[{"name":"Paul Beame","is_ca":false},{"name":"Faith E. Fich","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02422048326305505,"gpt":0.2379167837357343,"spread":0.2136963004726792,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01388675,0.004301832,0.005248293,0.006822004,0.003293706,0.01167747,0.009582475,0.005806073,0.02723122],"category_scores_gemma":[0.09448473,0.002694789,0.002723225,0.008747675,0.005521022,0.02683913,0.008437512,0.01354675,0.004042646],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00642968,"about_ca_system_score_gemma":0.005634132,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00272913,"about_ca_topic_score_gemma":0.004108267,"domain_scores_codex":[0.9897192,0.003596846,0.0004904506,0.001540284,0.002837656,0.001815678],"domain_scores_gemma":[0.880115,0.1039227,0.002937282,0.00586886,0.00457511,0.002580948],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001534568,0.0008729274,0.001264379,0.001716371,0.0001844043,0.0001836769,0.0005932372,0.1739486,0.00219366,0.5973677,0.05540515,0.1647352],"study_design_scores_gemma":[0.0001510327,0.000162539,0.0004079956,0.0003108114,0.0001235486,0.0002581633,0.0001767245,0.3416273,0.001315364,0.6475728,0.007824153,0.00006957492],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03910154,0.026489,0.8336782,0.01485309,0.001743528,0.0003798165,0.00158418,0.001327128,0.08084351],"genre_scores_gemma":[0.3878987,0.02418773,0.5282868,0.004502759,0.006756912,0.001541377,0.004102891,0.002489423,0.04023356],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02723122,"threshold_uncertainty_score":0.09109747,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2044014345","doi":"10.5555/1109557.1109599","title":"Rank/select operations on large alphabets: a tool for text indexing","year":2006,"lang":"en","type":"article","venue":"Symposium on Discrete Algorithms","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":190,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Search engine indexing; Rank (graph theory); Generalization; String (physics); Variety (cybernetics); Computer science; Representation (politics); Alphabet; Combinatorics; Binary number; Binary search algorithm; Theoretical computer science; Mathematics; Algorithm; Information retrieval; Search algorithm; Artificial intelligence; Arithmetic","authors":[{"name":"Alexander Golynski","is_ca":true},{"name":"J. Ian Munro","is_ca":true},{"name":"Srinivasa Rao Satti","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.008969900852929844,"gpt":0.2570800764681503,"spread":0.2481101756152204,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002363927,0.001021335,0.002107982,0.003032393,0.00180413,0.003434257,0.002587882,0.001668756,0.01217717],"category_scores_gemma":[0.01411651,0.0007839616,0.001193491,0.006319441,0.00231729,0.01096172,0.003957884,0.002788253,0.006487657],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008916512,"about_ca_system_score_gemma":0.001141498,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001078952,"about_ca_topic_score_gemma":0.001317115,"domain_scores_codex":[0.9970132,0.0007469373,0.0002934503,0.0004142309,0.001256807,0.0002754564],"domain_scores_gemma":[0.9888911,0.005473206,0.0008113441,0.003783085,0.0006370175,0.0004041524],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00143034,0.0003781518,0.0009315339,0.0005660021,0.00006503077,0.0004189219,0.0007347135,0.0274776,0.02722242,0.2494606,0.04653882,0.6447759],"study_design_scores_gemma":[0.0003437469,0.0005862836,0.0003558938,0.0001338802,0.00008465488,0.001066283,0.0003907521,0.3593096,0.04160917,0.5429303,0.05304059,0.000148868],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01379317,0.0005632254,0.971505,0.0008138574,0.0001359699,0.0001851048,0.0006931847,0.007931639,0.004378889],"genre_scores_gemma":[0.138229,0.0008171932,0.8487861,0.0005436876,0.0004914076,0.0005092109,0.001770376,0.0009471759,0.007905838],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01217717,"threshold_uncertainty_score":0.04073668,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2037968786","doi":"10.1016/s0022-0000(03)00078-3","title":"On the parameterized complexity of the fixed alphabet shortest common supersequence and longest common subsequence problems","year":2003,"lang":"en","type":"article","venue":"Journal of Computer and System Sciences","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":179,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"Southeast Fisheries Science Center","keywords":"Parameterized complexity; Combinatorics; Longest common subsequence problem; Alphabet; Mathematics; Sequence (biology); Subsequence; Discrete mathematics; Biology; Genetics","authors":[{"name":"Krzysztof Pietrzak","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05730906861657043,"gpt":0.2577674707612815,"spread":0.2004584021447111,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005130848,0.001767409,0.003587637,0.00195262,0.00260162,0.007427807,0.005469713,0.003579617,0.01502354],"category_scores_gemma":[0.05108801,0.001196134,0.002502336,0.005442751,0.003473654,0.02141277,0.003885397,0.005489084,0.001349185],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005740596,"about_ca_system_score_gemma":0.006525215,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006020892,"about_ca_topic_score_gemma":0.006233648,"domain_scores_codex":[0.9927238,0.002587775,0.0005173686,0.001491825,0.001624506,0.001054778],"domain_scores_gemma":[0.9297576,0.05829419,0.002782638,0.005622858,0.002186689,0.001355983],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002381964,0.0006945009,0.004952476,0.0008657182,0.0003593425,0.0003997753,0.0007780174,0.5859976,0.003186855,0.2854383,0.02277924,0.09216627],"study_design_scores_gemma":[0.0001917007,0.00009352026,0.0006455039,0.00004530969,0.00007467072,0.0001265829,0.0002137297,0.6025651,0.0008689812,0.393231,0.001907463,0.00003639222],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4366393,0.004648034,0.5007979,0.01446331,0.0005997009,0.0005808231,0.005798258,0.002166563,0.03430616],"genre_scores_gemma":[0.8142553,0.002262437,0.1627016,0.001055073,0.0009470225,0.0006897203,0.007301442,0.00106961,0.009717708],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01502354,"threshold_uncertainty_score":0.05025882,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2145195191","doi":"10.1002/spe.2325","title":"Better bitmap performance with Roaring bitmaps","year":2015,"lang":"en","type":"article","venue":"Software Practice and Experience","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":159,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Saint John Regional Hospital; Université TÉLUQ; Université du Québec à Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Bitmap; Computer science; Encoding (memory); Compression (physics); Oracle; Data compression; Compression ratio; Parallel computing; Algorithm; Computer graphics (images); Artificial intelligence; Engineering; Programming language","authors":[{"name":"Samy Chambi","is_ca":true},{"name":"Daniel Lemire","is_ca":true},{"name":"Owen Kaser","is_ca":true},{"name":"Robert Godin","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0231628126247053,"gpt":0.2617781121611502,"spread":0.2386152995364449,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001091028,0.000593595,0.0006722545,0.001660515,0.0004417954,0.002102176,0.001354155,0.0008377655,0.01085815],"category_scores_gemma":[0.01073195,0.0002486899,0.000333343,0.00448532,0.0006403158,0.005871381,0.001306755,0.0009140596,0.002738688],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006715572,"about_ca_system_score_gemma":0.0006412764,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00217179,"about_ca_topic_score_gemma":0.001492016,"domain_scores_codex":[0.9986644,0.0001799352,0.0001445132,0.0001904346,0.0006330838,0.0001876348],"domain_scores_gemma":[0.9931239,0.002613682,0.0003860852,0.002185468,0.001509833,0.0001809738],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00357494,0.0005969531,0.005478791,0.0007147207,0.0001483134,0.0005448783,0.0007522617,0.07788181,0.1006262,0.02606049,0.04818112,0.7354396],"study_design_scores_gemma":[0.0003356458,0.001685996,0.005172495,0.0002338119,0.0001070543,0.001272199,0.0009000192,0.5373897,0.3621564,0.02067437,0.06979262,0.0002797222],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7573051,0.00693309,0.1618399,0.002452776,0.0009619512,0.0002268426,0.003112277,0.03593791,0.03123014],"genre_scores_gemma":[0.8389448,0.001139705,0.1444491,0.0006381925,0.000161933,0.0001343848,0.006130544,0.001466801,0.006934623],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01085815,"threshold_uncertainty_score":0.03632414,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2111044311","doi":"10.1093/bioinformatics/bts593","title":"SCALCE: boosting sequence compression algorithms using locally consistent encoding","year":2012,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":158,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"National Human Genome Research Institute; Natural Sciences and Engineering Research Council of Canada; ExxonMobil Research and Engineering Company","keywords":"Boosting (machine learning); Encoding (memory); Computer science; Algorithm; Sequence (biology); Data compression; Compression (physics); Artificial intelligence","authors":[{"name":"Faraz Hach","is_ca":true},{"name":"Ibrahim Numanagić","is_ca":true},{"name":"Can Alkan","is_ca":true},{"name":"S. Cenk Şahinalp","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08296651646590367,"gpt":0.3066863161757942,"spread":0.2237197997098906,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001136066,0.0007434042,0.0007818696,0.001143777,0.0005325642,0.0008230582,0.00171114,0.0009251098,0.002977152],"category_scores_gemma":[0.003878002,0.0002731679,0.0005396414,0.001373934,0.0008825218,0.001326089,0.001243823,0.001645308,0.001835579],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007019218,"about_ca_system_score_gemma":0.00100774,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00252289,"about_ca_topic_score_gemma":0.002297467,"domain_scores_codex":[0.9993893,0.0001059662,0.00003702645,0.00009191695,0.0003089935,0.00006666286],"domain_scores_gemma":[0.9986547,0.0005528362,0.000110542,0.0002921071,0.0003269042,0.00006275735],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008025577,0.000223342,0.002414393,0.0002858059,0.00007268279,0.0003328726,0.0003101166,0.1743041,0.05288671,0.02911741,0.01880398,0.7204459],"study_design_scores_gemma":[0.00007370277,0.0002316163,0.0005238502,0.0000343046,0.00002061756,0.0002121663,0.00005285556,0.9380634,0.03338635,0.01794322,0.009420172,0.00003774903],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02815193,0.0007519269,0.9618449,0.000253575,0.0001531205,0.0001501579,0.0002382333,0.006206397,0.002249726],"genre_scores_gemma":[0.2109031,0.0005344778,0.7799087,0.0006136544,0.0001737837,0.0003101907,0.001616393,0.0007584445,0.005181306],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002977152,"threshold_uncertainty_score":0.009959579,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2053550438","doi":"10.5555/338219.338634","title":"Adaptive set intersections, unions, and differences","year":2000,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":157,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of New Brunswick; University of Waterloo","funders":"","keywords":"Set (abstract data type); Computer science; Programming language","authors":[{"name":"Erik D. Demaine","is_ca":true},{"name":"Alejandro López-Ortíz","is_ca":true},{"name":"J. Ian Munro","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02080502840567923,"gpt":0.2353265507327696,"spread":0.2145215223270904,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005237977,0.0007578076,0.001219922,0.002513789,0.001557132,0.004532402,0.003460883,0.002464211,0.004821366],"category_scores_gemma":[0.04286161,0.0009684744,0.001281746,0.00451704,0.006094436,0.01951046,0.005345464,0.003670098,0.0008282373],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001922434,"about_ca_system_score_gemma":0.001316543,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008646449,"about_ca_topic_score_gemma":0.000699007,"domain_scores_codex":[0.9927344,0.002093451,0.0004698645,0.001863239,0.002407435,0.0004316071],"domain_scores_gemma":[0.9602599,0.03032435,0.003321032,0.004127795,0.001357375,0.0006094886],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006734694,0.0002044025,0.00572636,0.0004854138,0.00008803668,0.0001696105,0.0008798385,0.09499747,0.005500189,0.6638961,0.005217377,0.2221617],"study_design_scores_gemma":[0.00005658692,0.000196997,0.0009684798,0.00006456183,0.00004953386,0.000469819,0.0003539356,0.2736309,0.009045899,0.708169,0.006939049,0.00005518671],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.08274144,0.002149504,0.903833,0.001826551,0.0001165509,0.0002153906,0.0002664406,0.000840916,0.008010177],"genre_scores_gemma":[0.4966994,0.001218972,0.4971362,0.0004148235,0.0003005155,0.0003018642,0.0005185099,0.0002655577,0.003144111],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005237977,"threshold_uncertainty_score":0.02770144,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2109692166","doi":"10.1016/s0166-218x(03)00382-2","title":"On spaced seeds for similarity search","year":2003,"lang":"en","type":"article","venue":"Discrete Applied Mathematics","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":142,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Western University","funders":"","keywords":"Similarity (geometry); Mathematics; Sensitivity (control systems); Combinatorics; Nearest neighbor search; Algorithm; Computer science; Data mining; Artificial intelligence","authors":[{"name":"Uri Keich","is_ca":false},{"name":"Ming Li","is_ca":false},{"name":"Bin Ma","is_ca":true},{"name":"John Tromp","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02766122019691645,"gpt":0.2810312807924654,"spread":0.253370060595549,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003113815,0.0008985958,0.002287477,0.003116174,0.001253168,0.001889827,0.002648154,0.003115512,0.006147393],"category_scores_gemma":[0.02936116,0.0008859016,0.0007806519,0.004602799,0.002658436,0.00588033,0.004109342,0.002017531,0.001977602],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009531021,"about_ca_system_score_gemma":0.001121723,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001353115,"about_ca_topic_score_gemma":0.001377075,"domain_scores_codex":[0.9971868,0.001055863,0.0001704656,0.0004622711,0.0009745427,0.0001500619],"domain_scores_gemma":[0.9876727,0.007969343,0.0005388933,0.002373751,0.001040323,0.0004048739],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001135343,0.0001904674,0.001016659,0.0002722408,0.00007719654,0.0002501874,0.0003653315,0.1922906,0.01081723,0.4386129,0.00857016,0.3464017],"study_design_scores_gemma":[0.00009602825,0.0001248928,0.0001810952,0.00004221769,0.00002048121,0.0001968559,0.00005176468,0.6872347,0.002816795,0.3054889,0.003720426,0.00002577474],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01135403,0.0006953711,0.9850773,0.0002163584,0.0001019619,0.00005569567,0.00005312582,0.0003526721,0.002093425],"genre_scores_gemma":[0.2690101,0.0008481434,0.7224017,0.0002841293,0.0003144557,0.0002043938,0.0004604425,0.000342141,0.006134434],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006147393,"threshold_uncertainty_score":0.02056503,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2395489635","doi":"10.5555/2627817.2627898","title":"Dynamic graph connectivity in polylogarithmic worst case time","year":2013,"lang":"en","type":"article","venue":"Symposium on Discrete Algorithms","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":141,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Victoria","funders":"","keywords":"Combinatorics; Time complexity; Binary logarithm; Amortized analysis; Computer science; Graph; Enhanced Data Rates for GSM Evolution; Preprocessor; Upper and lower bounds; Data structure; Matching (statistics); Discrete mathematics; Path (computing); Sequence (biology); Mathematics; Algorithm","authors":[{"name":"Bruce M. Kapron","is_ca":true},{"name":"Valerie King","is_ca":true},{"name":"Ben Mountjoy","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.006188313789722419,"gpt":0.2324596412625818,"spread":0.2262713274728593,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004080792,0.002595172,0.002303372,0.002347942,0.002249417,0.007816906,0.005159807,0.003523184,0.01337869],"category_scores_gemma":[0.03013197,0.001692713,0.002158162,0.006499964,0.003049057,0.01665812,0.00383205,0.004548382,0.00316349],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006602376,"about_ca_system_score_gemma":0.006585269,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006565151,"about_ca_topic_score_gemma":0.01251666,"domain_scores_codex":[0.9867727,0.002207658,0.0008988484,0.003066849,0.004694249,0.002359725],"domain_scores_gemma":[0.9613441,0.02590123,0.002592353,0.00735708,0.001795588,0.00100964],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.004730037,0.00111542,0.008050062,0.001990333,0.0005228723,0.0009491391,0.00114897,0.4652947,0.03721242,0.0889868,0.0668744,0.3231249],"study_design_scores_gemma":[0.0007033346,0.0002883788,0.001412825,0.0001267993,0.0002802556,0.001072372,0.0004795809,0.7546094,0.01428249,0.2114932,0.01516393,0.00008742775],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2107775,0.00545556,0.6868343,0.02028388,0.0008978042,0.001075591,0.008608968,0.02410525,0.04196108],"genre_scores_gemma":[0.6214911,0.001486889,0.3544714,0.002025123,0.000608467,0.001013113,0.006820535,0.002304437,0.00977897],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01337869,"threshold_uncertainty_score":0.04790384,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2053743126","doi":"10.1006/jcss.2002.1823","title":"Finding Similar Regions in Many Sequences","year":2002,"lang":"en","type":"article","venue":"Journal of Computer and System Sciences","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":139,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Western University","funders":"","keywords":"Combinatorics; Polynomial-time approximation scheme; Mathematics; Hamming distance; Core (optical fiber); Sequence (biology); Time complexity; Approximation algorithm; Discrete mathematics; Entropy (arrow of time); Computer science; Biology; Physics; Genetics","authors":[{"name":"Ming Li","is_ca":false},{"name":"Bin Ma","is_ca":true},{"name":"Lusheng Wang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05043055801681404,"gpt":0.2647240296950133,"spread":0.2142934716781993,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005125258,0.0007232936,0.001198863,0.004316458,0.001382271,0.001037173,0.001012748,0.00176686,0.003174988],"category_scores_gemma":[0.004293627,0.000474742,0.0007319643,0.00380274,0.000831756,0.001874744,0.001083068,0.001007242,0.001327049],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003455141,"about_ca_system_score_gemma":0.000679385,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000844578,"about_ca_topic_score_gemma":0.001570126,"domain_scores_codex":[0.9990262,0.0001064009,0.000103033,0.000279273,0.000391127,0.00009397869],"domain_scores_gemma":[0.9974246,0.001049078,0.0004109656,0.0004766417,0.0004592061,0.0001796205],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002873335,0.0008541502,0.02799882,0.00104456,0.0004540601,0.005900218,0.001175968,0.02216006,0.3670491,0.01235736,0.005962624,0.5521697],"study_design_scores_gemma":[0.0004440208,0.003313964,0.04899443,0.0003734583,0.0011811,0.02172153,0.003178627,0.5418472,0.2659518,0.07395004,0.03881861,0.0002252262],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6998252,0.003955237,0.2889021,0.0009148869,0.0004007849,0.0002548776,0.001002227,0.00135029,0.003394505],"genre_scores_gemma":[0.7731761,0.00125896,0.2173216,0.0003815051,0.0004200424,0.0001472296,0.002934315,0.0002041612,0.004156094],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004316458,"threshold_uncertainty_score":0.01062143,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2889833303","doi":"10.1145/3375890","title":"Fully Functional Suffix Trees and Optimal Text Searching in BWT-Runs Bounded Space","year":2020,"lang":"en","type":"article","venue":"Journal of the ACM","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":138,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"","keywords":"Log-log plot; Binary logarithm; Bounded function; Search engine indexing; Space (punctuation); Suffix; Generalized suffix tree; Suffix tree; Suffix array","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.03092524817749018,"gpt":0.2482270717439886,"spread":0.2173018235664985,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001732215,0.0008935678,0.001586829,0.001717213,0.001153439,0.003406618,0.003032583,0.001743111,0.004700832],"category_scores_gemma":[0.01432714,0.0007219883,0.001149789,0.004783262,0.002109254,0.01036245,0.0033332,0.001988992,0.003137096],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001715449,"about_ca_system_score_gemma":0.002210414,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002688951,"about_ca_topic_score_gemma":0.003365817,"domain_scores_codex":[0.9963297,0.0007564528,0.0004270785,0.0008924361,0.00112105,0.0004732653],"domain_scores_gemma":[0.9916945,0.004156969,0.0005669447,0.002559929,0.0007775444,0.0002441314],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001993994,0.0003571077,0.002572204,0.0008752211,0.0001066619,0.0005653216,0.001375588,0.1630817,0.0398444,0.2976097,0.02348971,0.4681285],"study_design_scores_gemma":[0.0001867942,0.0002694005,0.0005494927,0.0001042275,0.00005400207,0.0004841162,0.0002311254,0.5128637,0.01228833,0.4580338,0.01485999,0.00007506116],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1200477,0.002688652,0.8435944,0.001779389,0.000224579,0.0001714184,0.002134474,0.01404987,0.01530939],"genre_scores_gemma":[0.329333,0.0009796878,0.6566136,0.0004936987,0.0002794515,0.000364066,0.003968506,0.001387987,0.006580095],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004700832,"threshold_uncertainty_score":0.01572585,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2991376628","doi":"10.1016/j.physrep.2019.09.005","title":"Data science applications to string theory","year":2019,"lang":"en","type":"article","venue":"Physics Reports","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":135,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council; Banff International Research Station for Mathematical Innovation and Discovery; Universidad Nacional Autónoma de México; Aspen Center for Physics; Abdus Salam International Centre for Theoretical Physics; CERN; University of Pennsylvania; Microsoft Research","keywords":"String (physics); Physics; Cluster analysis; Variety (cybernetics); Machine learning; String theory; Unsupervised learning; Artificial intelligence; Theoretical computer science; Computer science; Theoretical physics","authors":[{"name":"Fabian Ruehle","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03986754876128853,"gpt":0.3207699015752767,"spread":0.2809023528139882,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004134838,0.001394868,0.001601306,0.006568644,0.001364699,0.004544041,0.00187827,0.003157096,0.007487623],"category_scores_gemma":[0.01794965,0.0007629748,0.002057031,0.008700429,0.004026658,0.006878457,0.003246201,0.007002025,0.003565715],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001872957,"about_ca_system_score_gemma":0.001308753,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009508238,"about_ca_topic_score_gemma":0.0005208731,"domain_scores_codex":[0.9956721,0.001631906,0.00059439,0.0005189915,0.00144422,0.0001382847],"domain_scores_gemma":[0.9877808,0.009297025,0.0004546357,0.001330608,0.0009935271,0.0001434115],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00001578985,0.00003494123,0.0003441692,0.001021745,0.00004438656,0.0001717989,0.0002262943,0.004597068,0.0006825959,0.8832139,0.01361591,0.09603138],"study_design_scores_gemma":[0.000006149175,0.0000287884,0.0002259443,0.0005006394,0.00001318773,0.0004563621,0.0000612194,0.01230782,0.0008931069,0.8303249,0.1551451,0.00003679171],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00176708,0.06364042,0.8990952,0.009513474,0.001945006,0.0001733815,0.001141383,0.0004814058,0.02224262],"genre_scores_gemma":[0.04326252,0.1157447,0.8161538,0.005475886,0.006375118,0.0009904486,0.001908443,0.0003804817,0.009708684],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007487623,"threshold_uncertainty_score":0.02504861,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2039944676","doi":"10.1109/tpami.2013.28","title":"Multi-Exemplar Affinity Propagation","year":2013,"lang":"en","type":"article","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":122,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Concordia University","funders":"University of Illinois at Chicago; North China University of Technology; National Natural Science Foundation of China","keywords":"Computer science; Artificial intelligence; Pattern recognition (psychology); Computer vision","authors":[{"name":"Changdong Wang","is_ca":false},{"name":"Jian-Huang Lai","is_ca":false},{"name":"Ching Y. Suen","is_ca":true},{"name":"Junyong Zhu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02480372966182174,"gpt":0.2668883885277947,"spread":0.242084658865973,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002319825,0.001745021,0.002002307,0.002873082,0.001581355,0.002079322,0.005700132,0.003111545,0.006354782],"category_scores_gemma":[0.009562515,0.001015399,0.001696713,0.004196075,0.001237958,0.003244054,0.003353761,0.002821996,0.003334927],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001392745,"about_ca_system_score_gemma":0.001376403,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007304807,"about_ca_topic_score_gemma":0.008443192,"domain_scores_codex":[0.997017,0.0005359724,0.000182614,0.0007207769,0.001269652,0.000274145],"domain_scores_gemma":[0.9943772,0.001719693,0.0003127217,0.0009984779,0.002426299,0.0001655864],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002075715,0.0002235733,0.001810057,0.0003111315,0.0002482167,0.0001834142,0.0002875725,0.3575767,0.007983476,0.02386585,0.01393077,0.5933716],"study_design_scores_gemma":[0.00001350034,0.00002916714,0.000210137,0.00001354119,0.00002079847,0.00008382134,0.00002941598,0.9828462,0.00378297,0.01046683,0.002486457,0.00001709953],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00485852,0.0002168697,0.9921733,0.000122375,0.00006951267,0.00007848025,0.00008392997,0.0008931283,0.001503809],"genre_scores_gemma":[0.2197856,0.0005425429,0.7663574,0.0005041488,0.0001887353,0.0003766709,0.001066668,0.0004490477,0.01072919],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007304807,"threshold_uncertainty_score":0.02125883,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2007979101","doi":"10.1016/s0022-0000(03)00075-8","title":"Solving large FPT problems on coarse-grained parallel machines","year":2003,"lang":"en","type":"article","venue":"Journal of Computer and System Sciences","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":114,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Victoria; Dalhousie University; Carleton University","funders":"","keywords":"Computer science; Vertex cover; Parallelism (grammar); Bounded function; Vertex (graph theory); Implementation; Cover (algebra); Sequence (biology); Parallel computing; Tree (set theory); Parallel algorithm; Algorithm; Theoretical computer science; Mathematics; Graph; Combinatorics; Approximation algorithm; Programming language","authors":[{"name":"James J. Cheetham","is_ca":true},{"name":"Frank Dehne","is_ca":true},{"name":"Andrew Rau‐Chaplin","is_ca":true},{"name":"Ulrike Stege","is_ca":true},{"name":"Peter J. Taillon","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01946566761545364,"gpt":0.2503981644689971,"spread":0.2309324968535434,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001624984,0.0008682701,0.001556583,0.0007989366,0.001358672,0.001811013,0.001666481,0.001727907,0.005499913],"category_scores_gemma":[0.01025569,0.0006660533,0.0008166677,0.001681596,0.001468045,0.003415003,0.001297601,0.00207283,0.0006886366],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001095068,"about_ca_system_score_gemma":0.001760738,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005860194,"about_ca_topic_score_gemma":0.007488388,"domain_scores_codex":[0.9989623,0.0002273639,0.00009119409,0.0002346846,0.0002749205,0.0002094413],"domain_scores_gemma":[0.9924758,0.005810309,0.0002649382,0.0008857927,0.0003809583,0.0001821682],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005478986,0.0002415232,0.002088162,0.0004613429,0.0001094063,0.0003984853,0.0002238305,0.7973028,0.006186997,0.0375796,0.01263789,0.1422221],"study_design_scores_gemma":[0.0001017018,0.00004377378,0.0002464396,0.00001020512,0.0000148545,0.00005358317,0.00006230189,0.9387869,0.002349291,0.05723299,0.001090294,0.000007719344],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4063807,0.002262314,0.5647924,0.00361778,0.0006678279,0.0001954641,0.0004967668,0.004907414,0.01667931],"genre_scores_gemma":[0.6324877,0.0004414269,0.3607965,0.0002975767,0.0002520003,0.0002308543,0.0006558288,0.0003605977,0.004477523],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005860194,"threshold_uncertainty_score":0.018399,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2538355508","doi":"10.1038/nmeth.4037","title":"Comparison of high-throughput sequencing data compression tools","year":2016,"lang":"en","type":"article","venue":"Nature Methods","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":114,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Throughput; Computer science; Benchmarking; Benchmark (surveying); Raw data; Data compression; Data set; DNA sequencing; Data mining; Set (abstract data type); Biology; Artificial intelligence; Operating system","authors":[{"name":"Ibrahim Numanagić","is_ca":true},{"name":"James Bonfield","is_ca":false},{"name":"Faraz Hach","is_ca":true},{"name":"Jan Voges","is_ca":false},{"name":"Jörn Östermann","is_ca":false},{"name":"Claudio Alberti","is_ca":false},{"name":"Marco Mattavelli","is_ca":false},{"name":"S. Cenk Şahinalp","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1614176675957748,"gpt":0.4801760690923179,"spread":0.3187584014965431,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006209938,0.001219767,0.000948347,0.003917733,0.000777176,0.002280289,0.001979571,0.001410725,0.003659627],"category_scores_gemma":[0.02357206,0.0005153536,0.001002294,0.004154898,0.000559311,0.0021404,0.000996041,0.001088243,0.001389969],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001141872,"about_ca_system_score_gemma":0.001494321,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001296501,"about_ca_topic_score_gemma":0.00138007,"domain_scores_codex":[0.9957175,0.0008810699,0.0005105743,0.0004467567,0.002213574,0.0002305774],"domain_scores_gemma":[0.9817107,0.01156197,0.0007034947,0.001754979,0.003962076,0.0003068195],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.006605809,0.001079648,0.01371658,0.001991258,0.0006963049,0.0005454808,0.0005716322,0.08548509,0.1375825,0.01284079,0.01398408,0.7249008],"study_design_scores_gemma":[0.0004337538,0.001758645,0.01744241,0.0003351515,0.0003734433,0.0012248,0.0004368345,0.5890589,0.3583408,0.01053396,0.01976531,0.0002959987],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4339515,0.007536049,0.5144504,0.001380578,0.0009373322,0.0007242431,0.005651392,0.02677994,0.008588707],"genre_scores_gemma":[0.4106236,0.003272119,0.5680375,0.0004204992,0.0001896404,0.0007321677,0.01198735,0.001687251,0.003049819],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006209938,"threshold_uncertainty_score":0.03284168,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1995913408","doi":"10.1145/1989323.1989405","title":"Efficient diversity-aware search","year":2011,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":110,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Diversity (politics); Political science","authors":[{"name":"Albert Angel","is_ca":true},{"name":"Nick Koudas","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05953039391184662,"gpt":0.2397432936762116,"spread":0.180212899764365,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001289201,0.0007303726,0.001778739,0.001927261,0.0009470563,0.001821586,0.001649133,0.00153639,0.002674924],"category_scores_gemma":[0.008033267,0.0004874957,0.0005927766,0.003326651,0.0007810917,0.003524384,0.002449306,0.0009467004,0.001156143],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006830329,"about_ca_system_score_gemma":0.001342907,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001062407,"about_ca_topic_score_gemma":0.002037389,"domain_scores_codex":[0.9978497,0.0006478159,0.0001306585,0.0003582713,0.0007699764,0.0002436825],"domain_scores_gemma":[0.9963202,0.002034128,0.000202939,0.000895026,0.0004305258,0.0001170166],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007691761,0.0003170896,0.002448861,0.000418714,0.0001694139,0.0003330974,0.0003898885,0.322056,0.02642584,0.05856268,0.01517876,0.5729305],"study_design_scores_gemma":[0.00007741768,0.0001628553,0.0004818731,0.00002242259,0.00004882169,0.0004495785,0.0001107397,0.9181138,0.004755594,0.07088101,0.004869319,0.00002665139],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06213974,0.002854906,0.9253994,0.0005326105,0.00008241242,0.0001207969,0.0003039973,0.001074406,0.007491696],"genre_scores_gemma":[0.6315117,0.001213082,0.3605227,0.0002444928,0.0002328675,0.0001515079,0.0008164775,0.0001664221,0.005140742],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002674924,"threshold_uncertainty_score":0.008948565,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2101637789","doi":"10.1109/26.939851","title":"Single parity check product codes","year":2001,"lang":"en","type":"article","venue":"IEEE Transactions on Communications","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":110,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science","authors":[{"name":"David M. Rankin","is_ca":false},{"name":"T. Aaron Gulliver","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06367548601671887,"gpt":0.2947418912688785,"spread":0.2310664052521596,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001194381,0.0007215146,0.0006986594,0.0007930014,0.000525396,0.001340736,0.0009508673,0.0008111611,0.002131343],"category_scores_gemma":[0.006423314,0.0002341097,0.0003179044,0.0009885546,0.000896124,0.001421414,0.0008561082,0.0006724647,0.0007709817],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005899382,"about_ca_system_score_gemma":0.0008637432,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008717268,"about_ca_topic_score_gemma":0.0005416782,"domain_scores_codex":[0.9985777,0.0003859706,0.00004750602,0.0002005458,0.0006149263,0.0001732575],"domain_scores_gemma":[0.995749,0.002172621,0.000339593,0.0005607418,0.001084631,0.00009340876],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006994433,0.00008545083,0.002994537,0.0003837409,0.0001313274,0.0008431298,0.0001574511,0.4810214,0.0254542,0.307894,0.003675417,0.1766598],"study_design_scores_gemma":[0.00004356169,0.0002843422,0.0004316778,0.00004402477,0.00003836381,0.0009216766,0.00002344113,0.9014005,0.01721747,0.07529222,0.004260328,0.00004240096],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.155909,0.002829562,0.821845,0.0003968361,0.0001943672,0.0001071565,0.0003361137,0.0008802431,0.01750169],"genre_scores_gemma":[0.9012874,0.0008722052,0.09410259,0.0001410311,0.00008956116,0.00008373069,0.0002380766,0.00006591481,0.003119501],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002131343,"threshold_uncertainty_score":0.007130027,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1480780803","doi":"10.1109/dcc.1997.581951","title":"Linear-time, incremental hierarchy inference for compression","year":2002,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Computer science; Compression (physics); Data compression; Hierarchy; Sequence (biology); Paraphrase; Compression ratio; Artificial intelligence; Algorithm; Data compression ratio; Image compression; Theoretical computer science; Image (mathematics); Image processing","authors":[{"name":"Craig G. Nevill-Manning","is_ca":false},{"name":"Ian H. Witten","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03454588049591423,"gpt":0.27938073763913,"spread":0.2448348571432157,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001643297,0.001036901,0.001074266,0.001695157,0.001117505,0.001659384,0.003461384,0.001396392,0.007610481],"category_scores_gemma":[0.01472305,0.0006522611,0.0009805079,0.002270569,0.001982856,0.005940801,0.002650678,0.003030476,0.002291594],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002549446,"about_ca_system_score_gemma":0.002519663,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008257102,"about_ca_topic_score_gemma":0.01764042,"domain_scores_codex":[0.9982387,0.0004140096,0.0001172163,0.0003959176,0.0006565762,0.0001776065],"domain_scores_gemma":[0.9939238,0.003379898,0.0002410951,0.001715569,0.0005962148,0.0001433173],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000551195,0.0002385917,0.001565114,0.0005031733,0.00008918508,0.0002272777,0.0004154968,0.1328007,0.009010643,0.1413911,0.04769071,0.6655169],"study_design_scores_gemma":[0.00006444439,0.00006712561,0.0003023873,0.00004062244,0.00002823374,0.000137215,0.00006724088,0.8438805,0.006655441,0.1415834,0.007150004,0.00002341186],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01689335,0.001324369,0.9628123,0.001313323,0.0002044199,0.0001851887,0.0007036836,0.01011396,0.006449403],"genre_scores_gemma":[0.2378054,0.0005277466,0.7531076,0.0007182561,0.0003278888,0.0002917734,0.002263356,0.0006434593,0.004314495],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008257102,"threshold_uncertainty_score":0.02545959,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2114790712","doi":"10.1006/jagm.2000.1151","title":"Space Efficient Suffix Trees","year":2001,"lang":"en","type":"article","venue":"Journal of Algorithms","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":103,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Suffix; Space (punctuation); Computer science; Suffix tree; Artificial intelligence; Mathematics; Combinatorics; Linguistics; Philosophy","authors":[{"name":"Venkatesh Raman","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01461430931628228,"gpt":0.2550175803153401,"spread":0.2404032709990578,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008241601,0.000697427,0.001268633,0.001747667,0.001442472,0.00284235,0.001459757,0.001347143,0.0174628],"category_scores_gemma":[0.005875406,0.000613864,0.0008431487,0.004259832,0.0009091084,0.006270672,0.002498488,0.001747277,0.006416777],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007540916,"about_ca_system_score_gemma":0.001673265,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006067025,"about_ca_topic_score_gemma":0.001380691,"domain_scores_codex":[0.9982029,0.0003337337,0.0002183151,0.0002947679,0.0007789651,0.0001712748],"domain_scores_gemma":[0.9951782,0.001677964,0.0002082096,0.001920362,0.0008759894,0.0001392544],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0009675173,0.0004886262,0.001210693,0.0004105823,0.0001018713,0.0002257642,0.0003752421,0.02916099,0.03391625,0.1477174,0.06560559,0.7198195],"study_design_scores_gemma":[0.0002589003,0.0005413487,0.0008114811,0.0001549269,0.000168377,0.001173579,0.0003991628,0.3190011,0.0679945,0.5031661,0.1062588,0.00007174397],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.10424,0.004424814,0.8206254,0.00335613,0.00119092,0.0003445023,0.003800709,0.01251446,0.04950308],"genre_scores_gemma":[0.355516,0.002074019,0.586751,0.0008809536,0.0007027655,0.0003725487,0.009819175,0.00154022,0.04234314],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0174628,"threshold_uncertainty_score":0.05841887,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2147344024","doi":"10.1145/1841909.1841913","title":"Fast and Compact Web Graph Representations","year":2010,"lang":"en","type":"article","venue":"ACM Transactions on the Web","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":103,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Fondo Nacional de Desarrollo Científico y Tecnológico","keywords":"Computer science; Theoretical computer science; Graph","authors":[{"name":"Francisco Claude","is_ca":true},{"name":"Gonzalo Navarro","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01849260082025904,"gpt":0.2626085176005849,"spread":0.2441159167803258,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003227557,0.0006558992,0.0007662595,0.002573072,0.0003579207,0.001675491,0.001106834,0.0009561841,0.005722493],"category_scores_gemma":[0.004975036,0.0003739247,0.0004440581,0.003398582,0.0004387592,0.003678734,0.001607466,0.001172065,0.00231822],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004720906,"about_ca_system_score_gemma":0.0004659538,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001297406,"about_ca_topic_score_gemma":0.002249394,"domain_scores_codex":[0.9992837,0.0001117214,0.00004170057,0.0001023883,0.0003943875,0.00006610406],"domain_scores_gemma":[0.9980781,0.0005293273,0.0001276976,0.0008791671,0.0003370361,0.00004864512],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004313515,0.000153091,0.0009525148,0.0003869289,0.00006125337,0.0005992664,0.0003588146,0.1274365,0.02892245,0.06014583,0.0376167,0.7429352],"study_design_scores_gemma":[0.00007695084,0.0000869341,0.0007445607,0.00005332802,0.00003146209,0.0007456287,0.0002887545,0.8553949,0.02840684,0.09006534,0.02406881,0.00003645854],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07016766,0.001402289,0.9055863,0.0007895556,0.0002406897,0.0001620808,0.002454761,0.01058922,0.008607442],"genre_scores_gemma":[0.3848052,0.001452246,0.5948659,0.0003088783,0.0002060145,0.0002869779,0.007753261,0.00122426,0.009097269],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005722493,"threshold_uncertainty_score":0.01914364,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2130433404","doi":"10.1109/tit.2003.818411","title":"Efficient universal lossless data compression algorithms based on a greedy sequential grammar transform-part two: with context models","year":2003,"lang":"en","type":"article","venue":"IEEE Transactions on Information Theory","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":102,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Algorithm; Data compression; Lossless compression; Mathematics; Countable set; Entropy encoding; Arithmetic coding; Discrete mathematics; Theoretical computer science; Computer science; Context-adaptive binary arithmetic coding","authors":[{"name":"En‐hui Yang","is_ca":true},{"name":"Dake He","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02623472200375592,"gpt":0.242439027781114,"spread":0.2162043057773581,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00068594,0.0007808629,0.0007527643,0.0009214964,0.0004177271,0.0007165915,0.00130197,0.000888231,0.001023736],"category_scores_gemma":[0.002780165,0.0003190463,0.0006072923,0.001501455,0.0009607244,0.001654175,0.001600252,0.001232451,0.0005362043],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007056578,"about_ca_system_score_gemma":0.0008620404,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001035321,"about_ca_topic_score_gemma":0.001246448,"domain_scores_codex":[0.9995431,0.0001045135,0.00003313881,0.0001080484,0.0001666077,0.00004466209],"domain_scores_gemma":[0.9992993,0.0003653319,0.0000661368,0.0001533761,0.000089753,0.00002618106],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002191813,0.0001094587,0.000743768,0.0001817007,0.0000596119,0.0002165274,0.0003066789,0.1979513,0.03225037,0.142262,0.004235264,0.621464],"study_design_scores_gemma":[0.00002213065,0.00005912717,0.0001210735,0.00001373371,0.00001255908,0.000212888,0.00002898576,0.9460542,0.01324189,0.0381138,0.002105585,0.00001402356],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01145538,0.0004440444,0.9866229,0.0001172225,0.00002940775,0.00005184708,0.00004909204,0.0004448514,0.0007853],"genre_scores_gemma":[0.2016503,0.0005173312,0.795517,0.00015832,0.00005961966,0.0002066233,0.0002454277,0.00009681933,0.001548511],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00130197,"threshold_uncertainty_score":0.00511992,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2022117185","doi":"10.1145/509907.509950","title":"Cache-oblivious priority queue and graph algorithm applications","year":2002,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":101,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Priority queue; Cache; Parallel computing; Cache-oblivious algorithm; Queue; Cache algorithms; Graph; Theoretical computer science; CPU cache; Algorithm; Computer network","authors":[{"name":"Lars Arge","is_ca":false},{"name":"Michael A. Bender","is_ca":false},{"name":"Erik D. Demaine","is_ca":false},{"name":"Bryan Holland‐Minkley","is_ca":false},{"name":"J. Ian Munro","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01298618704765494,"gpt":0.2229243720543181,"spread":0.2099381850066632,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001340986,0.0005223607,0.0004706862,0.0008742869,0.0008488884,0.001702154,0.002301976,0.0007470887,0.003263378],"category_scores_gemma":[0.007583458,0.0003763506,0.0003963909,0.001625862,0.001070174,0.004034786,0.00155075,0.001799952,0.0007396328],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001968244,"about_ca_system_score_gemma":0.002496029,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00371842,"about_ca_topic_score_gemma":0.004904149,"domain_scores_codex":[0.9986911,0.0002960937,0.00008899024,0.0001814904,0.0005492392,0.0001931384],"domain_scores_gemma":[0.996437,0.001556308,0.0002541505,0.0008306662,0.0007821394,0.0001397479],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003829132,0.0002469574,0.001270012,0.0003290609,0.00005041349,0.0001471594,0.0003969725,0.1779525,0.01063302,0.5702517,0.01079989,0.2275394],"study_design_scores_gemma":[0.00006821497,0.0001426139,0.0001621284,0.00002908177,0.00003350121,0.0001402236,0.00005724272,0.7598951,0.01115669,0.2148451,0.01344739,0.00002275669],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02226317,0.000645413,0.9706302,0.0006724662,0.0001101187,0.00008486243,0.00008447219,0.0009701723,0.004539156],"genre_scores_gemma":[0.3101186,0.0009256782,0.6813584,0.0003888459,0.0001903608,0.000181505,0.0003471737,0.0002673468,0.006222126],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00371842,"threshold_uncertainty_score":0.01428074,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2124602574","doi":"10.1145/502512.502558","title":"Induction of semantic classes from natural language text","year":2001,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":99,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language processing; Artificial intelligence; Set (abstract data type); Task (project management); Space (punctuation); Natural language; Unsupervised learning; Information retrieval","authors":[{"name":"Dekang Lin","is_ca":true},{"name":"Patrick Pantel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01160689930725286,"gpt":0.2518886837407619,"spread":0.240281784433509,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001672562,0.0009821595,0.000749794,0.006785381,0.001742534,0.001380293,0.001352117,0.0008871856,0.002952703],"category_scores_gemma":[0.008946538,0.0004115948,0.001278647,0.003122626,0.001352201,0.003660142,0.001872757,0.001886307,0.001849902],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001204149,"about_ca_system_score_gemma":0.002346931,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001993676,"about_ca_topic_score_gemma":0.004000432,"domain_scores_codex":[0.997725,0.0005281327,0.0002031275,0.0007042468,0.0006990386,0.0001405317],"domain_scores_gemma":[0.9941854,0.003513986,0.0004864013,0.0005615124,0.001091661,0.0001610724],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002948279,0.0004482083,0.006577735,0.0006512877,0.00008358462,0.0002845388,0.001029004,0.006236929,0.01668105,0.03977891,0.02406508,0.9038689],"study_design_scores_gemma":[0.00016887,0.0002520733,0.01478467,0.0004791487,0.000206338,0.0009713718,0.00219603,0.4643527,0.06131,0.3263514,0.1287586,0.000168882],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05537136,0.0008557125,0.92278,0.001192693,0.0002773623,0.0008198968,0.004421255,0.005594963,0.008686723],"genre_scores_gemma":[0.1445165,0.0005455428,0.8332618,0.0002849956,0.0002250358,0.0008640752,0.01678769,0.0004227733,0.003091658],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006785381,"threshold_uncertainty_score":0.009877741,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2463091895","doi":"10.1093/bioinformatics/btw397","title":"ntHash: recursive nucleotide hashing","year":2016,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":98,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Canada's Michael Smith Genome Sciences Centre","funders":"National Human Genome Research Institute; National Institutes of Health","keywords":"Hash function; Computer science; Expediting; Software; Data mining; Sequence (biology); Universal hashing; Theoretical computer science; Hash table; Biology; Genetics; Programming language","authors":[{"name":"Hamid Mohamadi","is_ca":true},{"name":"Justin Chu","is_ca":true},{"name":"Benjamin P. Vandervalk","is_ca":true},{"name":"İnanç Birol","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01728474627139232,"gpt":0.2344576880311056,"spread":0.2171729417597133,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00122493,0.0006355576,0.000725003,0.0009525071,0.0006609383,0.001078385,0.002021729,0.0006327538,0.01113661],"category_scores_gemma":[0.005981087,0.0004059972,0.00058967,0.001461723,0.0008782407,0.001725978,0.002400193,0.0009290772,0.009234875],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005830812,"about_ca_system_score_gemma":0.001406157,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001549355,"about_ca_topic_score_gemma":0.001873056,"domain_scores_codex":[0.9986986,0.0002309792,0.0001271111,0.0002305833,0.0006078254,0.0001049451],"domain_scores_gemma":[0.9981304,0.0005490819,0.000143654,0.000607425,0.000461532,0.0001079325],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001615801,0.0001582889,0.00516848,0.0009871981,0.0001120953,0.0002598528,0.000460339,0.04329789,0.04504209,0.04136975,0.07508264,0.7864456],"study_design_scores_gemma":[0.0003688779,0.0009581667,0.003337289,0.0001689574,0.00009384479,0.001579977,0.0002537536,0.5846991,0.1550042,0.09393184,0.1593722,0.0002319022],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02733463,0.001514825,0.9328129,0.0003716009,0.0004521915,0.0003314487,0.002642813,0.02562112,0.00891849],"genre_scores_gemma":[0.2566374,0.0006735681,0.7153977,0.0004237905,0.0002433567,0.0005057827,0.009912255,0.002070465,0.01413574],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01113661,"threshold_uncertainty_score":0.0372557,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1553071595","doi":"10.3233/fi-2011-565","title":"Self-Indexed Grammar-Based Compression","year":2011,"lang":"en","type":"article","venue":"Fundamenta Informaticae","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":96,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo; Natural Sciences and Engineering Research Council of Canada","funders":"","keywords":"Computer science; Grammar; Compression (physics); Natural language processing; Linguistics; Materials science; Philosophy","authors":[{"name":"Francisco Claude","is_ca":true},{"name":"Gonzalo Navarro","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03041967066636761,"gpt":0.2243386128071558,"spread":0.1939189421407882,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004187757,0.0005699726,0.0006180342,0.001625562,0.000423991,0.001085575,0.001373502,0.0006184369,0.006291499],"category_scores_gemma":[0.002892423,0.0003041316,0.0005496937,0.002300026,0.000762684,0.002134655,0.001665428,0.000781142,0.002705408],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006801352,"about_ca_system_score_gemma":0.001138777,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001147186,"about_ca_topic_score_gemma":0.001225374,"domain_scores_codex":[0.9991122,0.00009194478,0.00009025502,0.0001632292,0.0004428479,0.0000995213],"domain_scores_gemma":[0.9981657,0.0004146512,0.00009111989,0.0007299236,0.0005523136,0.00004625753],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004629686,0.0002168871,0.001354172,0.0004436396,0.00006001705,0.000821639,0.000515415,0.03289752,0.1028585,0.08769705,0.0424288,0.7302435],"study_design_scores_gemma":[0.0002009336,0.0004009992,0.00168356,0.0001243395,0.00010211,0.002068917,0.0002384813,0.4844668,0.2705157,0.1228513,0.117229,0.0001178717],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06724194,0.001109832,0.8819641,0.0006033234,0.0003955403,0.0004105859,0.002888451,0.02283825,0.02254804],"genre_scores_gemma":[0.3359875,0.0009934411,0.6262264,0.0007206728,0.0002605786,0.0005700882,0.008795657,0.00297082,0.0234748],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006291499,"threshold_uncertainty_score":0.02104717,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2164030956","doi":"10.1109/dcc.2007.44","title":"High Throughput Compression of Double-Precision Floating-Point Data","year":2007,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":96,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Department of Family and Community Medicine, University of Toronto","keywords":"Lossless compression; Throughput; Computer science; Data compression; Compression (physics); Compression ratio; Point (geometry); Floating point; Computer hardware; Algorithm; Real-time computing; Wireless; Engineering; Mathematics; Telecommunications","authors":[{"name":"Martin Burtscher","is_ca":false},{"name":"Paruj Ratanaworabhan","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0562830034262234,"gpt":0.3208830209361637,"spread":0.2646000175099403,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007199424,0.0008237549,0.0006360144,0.002517278,0.0004607168,0.001048513,0.0009948338,0.0004727557,0.002101953],"category_scores_gemma":[0.00360263,0.0001988691,0.000258721,0.003415602,0.000360724,0.001738245,0.0009225524,0.0007390192,0.00130922],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000486934,"about_ca_system_score_gemma":0.0005535355,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001439301,"about_ca_topic_score_gemma":0.001166609,"domain_scores_codex":[0.9989461,0.00006222088,0.00007269128,0.0001158641,0.0007215271,0.00008164207],"domain_scores_gemma":[0.9986192,0.0003578356,0.00009338641,0.0002709518,0.0006148603,0.00004386933],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001026393,0.000157576,0.002473488,0.0004640825,0.00007978058,0.0008908573,0.000383435,0.02652965,0.1451492,0.006013861,0.03261606,0.7842156],"study_design_scores_gemma":[0.0002534761,0.0004276705,0.006620003,0.0001645035,0.00006119321,0.001766175,0.0002742534,0.3261798,0.5985149,0.01199306,0.05362663,0.0001183102],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2410059,0.004626178,0.7211755,0.000923691,0.0006808902,0.0004840004,0.00712942,0.01523023,0.008744183],"genre_scores_gemma":[0.5243355,0.002429566,0.4444966,0.0002777377,0.0003907428,0.000520121,0.01878969,0.001060902,0.007699154],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.002517278,"threshold_uncertainty_score":0.007031739,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2012680439","doi":"10.1109/51.940049","title":"A compression algorithm for DNA sequences","year":2001,"lang":"en","type":"article","venue":"IEEE Engineering in Medicine and Biology Magazine","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":93,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; University of Hong Kong; City University of Hong Kong","keywords":"Data compression; Compression (physics); Algorithm; Benchmark (surveying); Computer science; Matching (statistics); DNA sequencing; DNA; DNA computing; Mathematics; Biology; Genetics; Physics; Computation","authors":[{"name":"Xin Chen","is_ca":false},{"name":"Sam Kwong","is_ca":false},{"name":"Ming Li","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03680986124003464,"gpt":0.2978930809314662,"spread":0.2610832196914316,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005337106,0.0009762071,0.000643562,0.001911051,0.0006964838,0.0009750935,0.0009314274,0.001256578,0.004106347],"category_scores_gemma":[0.003101591,0.0003256897,0.0005126274,0.002646521,0.0006622931,0.001434162,0.0009200164,0.0009654854,0.002018429],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005459511,"about_ca_system_score_gemma":0.0007675677,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001399163,"about_ca_topic_score_gemma":0.00108658,"domain_scores_codex":[0.999193,0.00008289819,0.00006699147,0.0001341918,0.0004715549,0.00005139339],"domain_scores_gemma":[0.9991628,0.0002639149,0.00006762671,0.0001968124,0.0002812853,0.00002749959],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000259745,0.00005812765,0.0004766479,0.0002333596,0.00004482672,0.0002081197,0.0001291938,0.03197046,0.04499648,0.02249699,0.010392,0.888734],"study_design_scores_gemma":[0.0001793662,0.0005969045,0.00154379,0.000174031,0.00009581251,0.003960522,0.0001303376,0.6654543,0.2089555,0.03963121,0.07917127,0.0001069701],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01095981,0.001206713,0.9824101,0.0002353662,0.0002208199,0.0001770209,0.0002823606,0.001997418,0.002510438],"genre_scores_gemma":[0.05532803,0.000898138,0.9367779,0.0002606962,0.0001475854,0.0003231622,0.001075045,0.0002082592,0.004981156],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004106347,"threshold_uncertainty_score":0.01373708,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2905575949","doi":"10.1093/bioinformatics/bty1015","title":"SPRING: a next-generation compressor for FASTQ data","year":2018,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":91,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"National Cancer Institute; University of Illinois at Urbana-Champaign; National Institutes of Health; Sunnybrook Research Institute; Silicon Valley Community Foundation","keywords":"Computer science; Lossless compression; Scalability; Lossy compression; Data compression; Data mining; Redundancy (engineering); Identifier; Compression (physics); Database; Artificial intelligence; Computer network; Operating system","authors":[{"name":"Shubham Chandak","is_ca":false},{"name":"Kedar Tatwawadi","is_ca":false},{"name":"Idoia Ochoa","is_ca":false},{"name":"Mikel Hernáez","is_ca":false},{"name":"Tsachy Weissman","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1701892097079504,"gpt":0.3126197987859841,"spread":0.1424305890780337,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002101411,0.00162748,0.0006736658,0.001727622,0.001282047,0.001512046,0.003015341,0.001190771,0.02184419],"category_scores_gemma":[0.008034871,0.0005950392,0.0008668639,0.002509367,0.001198091,0.003465723,0.002374373,0.00181802,0.009098105],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001069935,"about_ca_system_score_gemma":0.001629075,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003078847,"about_ca_topic_score_gemma":0.003242033,"domain_scores_codex":[0.9981984,0.0001615736,0.0001679425,0.0003119562,0.001014224,0.000145968],"domain_scores_gemma":[0.9966069,0.0009193017,0.0001896208,0.000920224,0.001223359,0.0001406303],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002930821,0.0003050132,0.004287962,0.001234687,0.000266272,0.0007021545,0.0007338182,0.01410287,0.06632976,0.02156779,0.3759215,0.5116174],"study_design_scores_gemma":[0.0009558438,0.000643967,0.00314173,0.0003235227,0.0001379192,0.001304596,0.0003212249,0.2697725,0.3561509,0.02458731,0.3423069,0.0003536525],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03717218,0.001932525,0.7046961,0.002084502,0.001280655,0.0009322666,0.0195895,0.2201468,0.01216541],"genre_scores_gemma":[0.1668357,0.00144597,0.7144243,0.001831378,0.0007037866,0.001995744,0.06849612,0.01438631,0.02988072],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02184419,"threshold_uncertainty_score":0.07307607,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2128020980","doi":"10.1145/1132516.1132540","title":"Optimal phylogenetic reconstruction","year":2006,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":90,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Harvard University; National Science Foundation","keywords":"Tree (set theory); Phylogenetic tree; Markov chain; Mathematics; Combinatorics; Algorithm; Tree rearrangement; Sequence (biology); Discrete mathematics; Biology; Statistics; Genetics","authors":[{"name":"Constantinos Daskalakis","is_ca":false},{"name":"Elchanan Mossel","is_ca":false},{"name":"Sébastien Roch","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.007480600463049214,"gpt":0.2068424045391989,"spread":0.1993618040761496,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00125014,0.0008301265,0.001405511,0.001725725,0.001050356,0.001486381,0.001466849,0.001927446,0.006307565],"category_scores_gemma":[0.01122188,0.0008040303,0.001016059,0.001862513,0.001309338,0.003871737,0.002245241,0.00191039,0.00229227],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001041145,"about_ca_system_score_gemma":0.001416166,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001040454,"about_ca_topic_score_gemma":0.001319652,"domain_scores_codex":[0.9984642,0.0003717908,0.0001142807,0.000490933,0.0003461794,0.0002126596],"domain_scores_gemma":[0.9957813,0.002128988,0.0002501038,0.001216772,0.0004822611,0.0001404696],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006789261,0.0001876672,0.003054662,0.0006016534,0.0001167378,0.000316448,0.0004990181,0.2092918,0.01686407,0.1904124,0.02070909,0.5572674],"study_design_scores_gemma":[0.00008980847,0.00009461533,0.0007993107,0.00006873179,0.000045569,0.0003540502,0.000212221,0.6167569,0.01026473,0.3589853,0.01228793,0.00004091243],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0332933,0.0007365197,0.9583474,0.0005800194,0.00008530971,0.00007009278,0.000577439,0.001423256,0.004886576],"genre_scores_gemma":[0.2496082,0.0007285018,0.7406767,0.0003048582,0.0001187022,0.0001620822,0.002622701,0.0004674604,0.005310812],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006307565,"threshold_uncertainty_score":0.02110094,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2048541619","doi":"10.1016/s0020-0190(02)00500-8","title":"Cuckoo hashing: Further analysis","year":2003,"lang":"en","type":"article","venue":"Information Processing Letters","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":89,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Cuckoo; Hash function; Cuckoo search; Theoretical computer science; Algorithm; Programming language","authors":[{"name":"Luc Devroye","is_ca":true},{"name":"Pat Morin","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.00865126449561512,"gpt":0.2204994529410758,"spread":0.2118481884454607,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001306641,0.0007424014,0.000990538,0.001816147,0.001409966,0.001909466,0.001567001,0.001083885,0.01772729],"category_scores_gemma":[0.01510229,0.0003534635,0.000688301,0.002860469,0.001672576,0.004268828,0.001583788,0.001670751,0.002328007],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001348098,"about_ca_system_score_gemma":0.001283648,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002354566,"about_ca_topic_score_gemma":0.002512161,"domain_scores_codex":[0.9986103,0.0003221039,0.00005117034,0.0001764425,0.0006570449,0.000182999],"domain_scores_gemma":[0.9935887,0.003143391,0.000352675,0.001472799,0.001261327,0.0001810233],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002766173,0.0001692318,0.003366688,0.000285048,0.00003650532,0.0002707605,0.0003583017,0.03372028,0.003066701,0.8250535,0.03002633,0.10337],"study_design_scores_gemma":[0.00003739777,0.00008142999,0.002790496,0.00007294877,0.00003310319,0.0006435518,0.00017637,0.4431173,0.002562916,0.5342103,0.01620326,0.0000708305],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1654066,0.006460902,0.7277422,0.005787178,0.001022536,0.0005553606,0.001156629,0.001393463,0.09047517],"genre_scores_gemma":[0.8656198,0.003198602,0.07894488,0.001058419,0.001441864,0.0003583964,0.001411232,0.0005773531,0.0473895],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01772729,"threshold_uncertainty_score":0.0593037,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2148659572","doi":"10.5555/1283383.1283456","title":"Succinct indexes for strings, binary relations and multi-labeled trees","year":2007,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":87,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Encoding (memory); String (physics); Set (abstract data type); Binary number; Rank (graph theory); Bit array; Type (biology); String searching algorithm; Data structure; Theoretical computer science; Combinatorics; Mathematics; Arithmetic","authors":[{"name":"Jérémy Barbay","is_ca":true},{"name":"Meng He","is_ca":true},{"name":"J. Ian Munro","is_ca":true},{"name":"Srinivasa Rao Satti","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02284381078697994,"gpt":0.281223494833351,"spread":0.258379684046371,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002095663,0.0009102777,0.0009598666,0.001720822,0.0008966262,0.003807002,0.002230764,0.00113939,0.00650267],"category_scores_gemma":[0.01078527,0.0007997246,0.001292675,0.003637153,0.002534258,0.01411545,0.003362529,0.002512316,0.002719959],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001904154,"about_ca_system_score_gemma":0.001994204,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001030065,"about_ca_topic_score_gemma":0.001626411,"domain_scores_codex":[0.9962246,0.0006111893,0.0006313592,0.0004966683,0.001771534,0.0002646479],"domain_scores_gemma":[0.9911431,0.002846981,0.001090809,0.003526773,0.001134625,0.0002576941],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000610331,0.0001613908,0.001531183,0.0007942631,0.00004938481,0.0003064259,0.0009041621,0.03713982,0.03500147,0.6542045,0.01381652,0.2554806],"study_design_scores_gemma":[0.0001265244,0.0004246345,0.0005287568,0.0003852392,0.0000933568,0.0006358508,0.0003149041,0.2425209,0.1213369,0.5121619,0.121275,0.0001960763],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006783208,0.0002920712,0.9871238,0.0002482765,0.00008417588,0.000141596,0.000715157,0.002006028,0.002605723],"genre_scores_gemma":[0.09138945,0.0006905411,0.8971453,0.0004671802,0.0001245856,0.0004909254,0.002580685,0.001047893,0.006063425],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00650267,"threshold_uncertainty_score":0.02175355,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2138523425","doi":"10.1109/dcc.1992.227475","title":"Constructing word-based text compression algorithms","year":2003,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":86,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo; University of Victoria","funders":"","keywords":"Huffman coding; Word (group theory); ASCII; Computer science; Context (archaeology); Algorithm; Compression (physics); Data compression; Alphanumeric; Sigma; Alphabet; Word problem (mathematics education); Compression ratio; Lossless compression; Theoretical computer science; Natural language processing; Artificial intelligence; Arithmetic; Mathematics; Programming language; Linguistics","authors":[{"name":"R. Nigel Horspool","is_ca":true},{"name":"Gordon V. Cormack","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01936819052923327,"gpt":0.2530895295810157,"spread":0.2337213390517824,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008133821,0.0008360915,0.0007024743,0.002816955,0.0006625699,0.001560177,0.001596576,0.001120476,0.005668475],"category_scores_gemma":[0.005498734,0.0004003473,0.0006674373,0.002616548,0.000674311,0.002075601,0.001554442,0.0009549614,0.00543822],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006360229,"about_ca_system_score_gemma":0.001004332,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007732165,"about_ca_topic_score_gemma":0.0006325783,"domain_scores_codex":[0.9989746,0.0001485863,0.0001386119,0.0001909927,0.0004577502,0.00008944165],"domain_scores_gemma":[0.9980865,0.0004301163,0.00009669725,0.0003046874,0.00102936,0.00005278227],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002134109,0.0001097352,0.0009462911,0.0003442329,0.00004957556,0.0002342513,0.0002857287,0.03346208,0.0365651,0.06221776,0.01105282,0.854519],"study_design_scores_gemma":[0.0001251609,0.0002831467,0.0007377084,0.0001384578,0.00009165685,0.0007484072,0.0002426254,0.6728899,0.1880181,0.0736453,0.06300986,0.00006969183],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01064252,0.0003341727,0.9819518,0.0001342619,0.000118598,0.0002921686,0.0002361581,0.002809478,0.003480941],"genre_scores_gemma":[0.04259743,0.0004390346,0.9516012,0.0001539965,0.00009439597,0.0003440141,0.0009950468,0.0003257598,0.003449106],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005668475,"threshold_uncertainty_score":0.01896292,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4254779780","doi":"10.1145/1109557.1109599","title":"Rank/select operations on large alphabets","year":2006,"lang":"en","type":"article","venue":"Proceedings of the seventeenth annual ACM-SIAM symposium on Discrete algorithm - SODA '06","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":86,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Search engine indexing; Rank (graph theory); Generalization; String (physics); Alphabet; Variety (cybernetics); Representation (politics); Computer science; Combinatorics; Binary number; Binary search algorithm; Theoretical computer science; Mathematics; Algorithm; Search algorithm; Arithmetic; Information retrieval; Artificial intelligence","authors":[{"name":"Alexander Golynski","is_ca":true},{"name":"J. Ian Munro","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.006358395880270913,"gpt":0.2351972356445886,"spread":0.2288388397643177,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001459467,0.000619885,0.001755974,0.00126584,0.001348392,0.002537076,0.00177156,0.001474533,0.01236642],"category_scores_gemma":[0.009729339,0.0004079432,0.00073848,0.003232681,0.001338231,0.008518698,0.00275747,0.001638666,0.003829368],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006898492,"about_ca_system_score_gemma":0.001090169,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001032096,"about_ca_topic_score_gemma":0.001462654,"domain_scores_codex":[0.9972264,0.000483817,0.0002501259,0.0004797947,0.001090592,0.0004692261],"domain_scores_gemma":[0.991407,0.00416594,0.0006882462,0.00281984,0.0005971694,0.0003217886],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003091218,0.0005978096,0.00227008,0.0008166234,0.00006490247,0.0009270312,0.0007937951,0.07352174,0.04260734,0.2157857,0.0404939,0.6190299],"study_design_scores_gemma":[0.0004283435,0.0008977134,0.000764027,0.0001136257,0.00008632644,0.001449868,0.0008940404,0.4678518,0.07605795,0.4109309,0.0403974,0.0001280305],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2301281,0.00129481,0.7352485,0.002422084,0.0003011656,0.0003582899,0.002374929,0.007302158,0.02056991],"genre_scores_gemma":[0.6165663,0.0008094101,0.3592553,0.0007935178,0.0005201496,0.000307497,0.002717926,0.0006841974,0.01834564],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01236642,"threshold_uncertainty_score":0.0413698,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1977986119","doi":"10.1016/j.tcs.2007.07.041","title":"Optimal lower bounds for rank and select indexes","year":2007,"lang":"en","type":"article","venue":"Theoretical Computer Science","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":85,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Eidgenössische Technische Hochschule Zürich","keywords":"Rank (graph theory); Data structure; Upper and lower bounds; Computer science; Word (group theory); Combinatorics; Binary logarithm; Space (punctuation); Index (typography); Order (exchange); Algorithm; Mathematics; Discrete mathematics","authors":[{"name":"Alexander Golynski","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.007316668029087797,"gpt":0.2601985053427479,"spread":0.2528818373136601,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01631416,0.006812792,0.007296803,0.01087173,0.003961644,0.01826018,0.009000503,0.006094439,0.02574082],"category_scores_gemma":[0.1021297,0.003071447,0.002546411,0.01382264,0.006767613,0.02295343,0.011351,0.0108116,0.007769041],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009002999,"about_ca_system_score_gemma":0.00962744,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002294021,"about_ca_topic_score_gemma":0.005243991,"domain_scores_codex":[0.97544,0.006419869,0.001144629,0.002092703,0.01058443,0.004318424],"domain_scores_gemma":[0.8910767,0.0842495,0.003703729,0.0117876,0.00628616,0.002896284],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002705854,0.0007409448,0.001755313,0.001510445,0.00025358,0.0001910012,0.0006986848,0.1426874,0.005925396,0.5695769,0.04872957,0.225225],"study_design_scores_gemma":[0.0002419067,0.0003184174,0.0007097223,0.0003201412,0.0002129251,0.0003676135,0.0002995322,0.372339,0.006349754,0.6064754,0.01224538,0.0001201752],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03202162,0.01836236,0.8754501,0.009395971,0.001074449,0.0004251556,0.003488718,0.003302239,0.05647928],"genre_scores_gemma":[0.3998834,0.0147456,0.5270165,0.003595388,0.005308063,0.001770717,0.005759505,0.003554504,0.03836636],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02574082,"threshold_uncertainty_score":0.08627856,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3124442824","doi":"","title":"SIMD Compression and the Intersection of Sorted Integers","year":2016,"lang":"en","type":"article","venue":"R-libre (Université Téluq)","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":84,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Université TÉLUQ","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"SIMD; Computer science; Parallel computing; Integer (computer science); Intersection (aeronautics); Compression (physics); Decoding methods; Data compression; State (computer science); Speedup; Compression ratio; Algorithm; Programming language","authors":[{"name":"Daniel Lemire","is_ca":true},{"name":"Leonid Boytsov","is_ca":false},{"name":"Nathan Kurz","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.003248471905147706,"gpt":0.1557511980705495,"spread":0.1525027261654018,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000863851,0.0006792218,0.0007037679,0.001924761,0.0006477109,0.001475803,0.00137123,0.0004549597,0.004070547],"category_scores_gemma":[0.004509033,0.0003394251,0.0004563183,0.004530752,0.001219208,0.003002187,0.001844264,0.0006405489,0.00150409],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001209119,"about_ca_system_score_gemma":0.001067997,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002401082,"about_ca_topic_score_gemma":0.002615933,"domain_scores_codex":[0.9980711,0.0002683681,0.0001762516,0.0002786184,0.001032004,0.0001737093],"domain_scores_gemma":[0.9977039,0.0008325937,0.0001811382,0.0007986674,0.0004384164,0.00004538891],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001782722,0.0002137325,0.004550603,0.0003177754,0.00008296674,0.0004074863,0.0006056195,0.04281569,0.07660984,0.07882006,0.01970565,0.7740878],"study_design_scores_gemma":[0.0002334364,0.0006652616,0.002608628,0.0001071087,0.000079589,0.001012675,0.0005090488,0.4930788,0.3666255,0.0789128,0.05605596,0.0001111731],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2566111,0.003106211,0.6945091,0.000838101,0.0002833282,0.0002550383,0.001644991,0.0173851,0.02536698],"genre_scores_gemma":[0.5752291,0.0009405708,0.4109557,0.0004002778,0.0001612308,0.0002773474,0.002792547,0.0009393791,0.008303851],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004070547,"threshold_uncertainty_score":0.01361734,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2077583313","doi":"10.1006/jagm.2000.1108","title":"Fast Algorithms to Generate Necklaces, Unlabeled Necklaces, and Irreducible Polynomials over GF(2)","year":2000,"lang":"en","type":"article","venue":"Journal of Algorithms","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":82,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Victoria","funders":"","keywords":"Mathematics; Combinatorics; Necklace; Alphabet; Permutation (music); Algorithm; Discrete mathematics","authors":[{"name":"Kevin Cattell","is_ca":true},{"name":"Frank Ruskey","is_ca":true},{"name":"Joe Sawada","is_ca":true},{"name":"M. Serra","is_ca":true},{"name":"C. Robert Miers","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01128066133810982,"gpt":0.2575280766633357,"spread":0.2462474153252258,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001152526,0.00109803,0.0007375249,0.001471089,0.001117126,0.001039796,0.001157211,0.0009726047,0.007803902],"category_scores_gemma":[0.005723136,0.0004869495,0.0007598053,0.001822785,0.0008536145,0.002204729,0.002535087,0.001475382,0.002003877],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0010048,"about_ca_system_score_gemma":0.001823123,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001452986,"about_ca_topic_score_gemma":0.004914508,"domain_scores_codex":[0.9989529,0.0002014271,0.00007215347,0.0001539195,0.000422865,0.000196678],"domain_scores_gemma":[0.9971402,0.001295128,0.0002684896,0.0006914715,0.0004848102,0.00011981],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001242211,0.0003825936,0.001541903,0.0004735614,0.00008044422,0.0002962636,0.0006327883,0.04463021,0.02591834,0.1178411,0.02171951,0.7852412],"study_design_scores_gemma":[0.0009496759,0.0007281257,0.0008835595,0.0001677561,0.0001378488,0.0007916269,0.0004927582,0.6070118,0.09397253,0.27161,0.02313145,0.0001228751],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06912894,0.0005438778,0.9179324,0.0004990304,0.0001483361,0.0004345244,0.0004328353,0.003998746,0.00688123],"genre_scores_gemma":[0.2607482,0.0003618125,0.7293639,0.0002113517,0.00009238197,0.0004285816,0.001465799,0.0003796687,0.006948301],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007803902,"threshold_uncertainty_score":0.0261066,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2106159865","doi":"10.1109/tmm.2003.822793","title":"Globally Optimal Uneven Error-Protected Packetization of Scalable Code Streams","year":2004,"lang":"en","type":"article","venue":"IEEE Transactions on Multimedia","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":82,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Erasure; Network packet; Algorithm; Scalability; Payload (computing); Binary logarithm; Discrete mathematics; Mathematics; Computer network","authors":[{"name":"Sorina Dumitrescu","is_ca":true},{"name":"Xiaobin Wu","is_ca":true},{"name":"Z. Wang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01642296285742713,"gpt":0.2544252547356321,"spread":0.238002291878205,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001015292,0.0007707466,0.0009241666,0.000480457,0.000392425,0.0007669741,0.001104517,0.0005520491,0.001583263],"category_scores_gemma":[0.004254486,0.0002926075,0.0004373995,0.0005848419,0.0008066238,0.001962744,0.001823083,0.001131095,0.0002806906],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000796204,"about_ca_system_score_gemma":0.0008405253,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001030522,"about_ca_topic_score_gemma":0.0006656218,"domain_scores_codex":[0.9993562,0.0001513812,0.00004033399,0.0001194396,0.0002358413,0.00009666877],"domain_scores_gemma":[0.99887,0.0005622035,0.0001281806,0.0002559109,0.0001378292,0.00004585544],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003002759,0.0000591081,0.0006382366,0.00009403429,0.00003534835,0.0001062617,0.0001652197,0.7000237,0.01418602,0.0437157,0.002024075,0.2386521],"study_design_scores_gemma":[0.00001388539,0.0000420456,0.00006464295,0.000006219531,0.00000435326,0.00003266418,0.00002709805,0.9788399,0.008332672,0.0118343,0.0007953988,0.000006780143],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01682608,0.00008183649,0.9818367,0.00005792405,0.00001417047,0.00003556421,0.00002426473,0.0003904174,0.0007329856],"genre_scores_gemma":[0.451381,0.000227755,0.5453911,0.00009750985,0.00003941585,0.0001418337,0.0002274932,0.0001893145,0.002304614],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.001583263,"threshold_uncertainty_score":0.005776882,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2133601898","doi":"10.1145/780542.780590","title":"A sublinear algorithm for weakly approximating edit distance","year":2003,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Edit distance; Substring; Sublinear function; Character (mathematics); Computer science; Upper and lower bounds; Algorithm; Time complexity; Combinatorics; Mathematics; Data structure","authors":[{"name":"Tuğkan Batu","is_ca":false},{"name":"Funda Ergün","is_ca":false},{"name":"Joe Kilian","is_ca":false},{"name":"Avner Magen","is_ca":true},{"name":"Sofya Raskhodnikova","is_ca":false},{"name":"Ronitt Rubinfeld","is_ca":false},{"name":"Rahul Sami","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01491292570024554,"gpt":0.2482037933123638,"spread":0.2332908676121183,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002989555,0.001978391,0.002730534,0.002805192,0.001342748,0.003462053,0.004427619,0.002312708,0.008640673],"category_scores_gemma":[0.03102971,0.0007908398,0.00178255,0.003809574,0.00156451,0.007437896,0.004083373,0.003688976,0.004991632],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002951204,"about_ca_system_score_gemma":0.00386544,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004867456,"about_ca_topic_score_gemma":0.005905678,"domain_scores_codex":[0.9926996,0.001668665,0.0007645261,0.001647467,0.002622242,0.0005975723],"domain_scores_gemma":[0.9784461,0.01235194,0.001101939,0.005728724,0.001892121,0.0004792109],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001482369,0.0005634854,0.006145979,0.0006229017,0.000230946,0.000233659,0.0005764993,0.111373,0.01642002,0.05847865,0.02147541,0.782397],"study_design_scores_gemma":[0.0002019123,0.000247588,0.0008840585,0.00003565099,0.00005973242,0.0004003785,0.0001368947,0.8696034,0.009901935,0.1122465,0.006232316,0.0000497896],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0245114,0.0005234668,0.9609815,0.0007048995,0.0001435776,0.0002942378,0.0006477899,0.007913107,0.004279986],"genre_scores_gemma":[0.2139343,0.0002094286,0.776901,0.0004515258,0.0001757367,0.0008319745,0.002507414,0.00078738,0.004201231],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008640673,"threshold_uncertainty_score":0.02890593,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}