{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":374,"total_is_capped":false,"direct_labels_cover":2,"predictions_cover":374,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"3de55248b53f","filters":{"topic":"Text and Document Classification Technologies"}},"results":[{"id":"W2170505850","doi":"10.1016/j.ipm.2009.03.002","title":"A systematic analysis of performance measures for classification tasks","year":2009,"lang":"en","type":"article","venue":"Information Processing & Management","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":6477,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Computer Research Institute of Montréal; Université de Montréal; Children's Hospital of Eastern Ontario","funders":"McGill University","keywords":"Confusion matrix; Confusion; Computer science; Artificial intelligence; Classifier (UML); Measure (data warehouse); Binary classification; Machine learning; Binary number; Natural language processing; Set (abstract data type); Class (philosophy); Data mining; Pattern recognition (psychology); Mathematics; Support vector machine; Arithmetic","authors":[{"name":"Marina Sokolova","is_ca":true},{"name":"Guy Lapalme","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02272805573408605,"gpt":0.2683619694755179,"spread":0.2456339137414319,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04296086,0.002308063,0.003285629,0.01642598,0.001276854,0.004170145,0.002322449,0.001661381,0.001398087],"category_scores_gemma":[0.2121844,0.0006902757,0.003032821,0.0144615,0.001154326,0.006485946,0.00166588,0.001895476,0.0008811959],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002928945,"about_ca_system_score_gemma":0.005282739,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0021972,"about_ca_topic_score_gemma":0.002573391,"domain_scores_codex":[0.9434702,0.02278243,0.009944817,0.005848186,0.01706079,0.0008935057],"domain_scores_gemma":[0.6392259,0.2589121,0.02155054,0.02473727,0.05435722,0.001216923],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.001123516,0.0007143526,0.03851762,0.008708969,0.002275045,0.00006682261,0.0004466531,0.007389662,0.007306534,0.004905843,0.005894669,0.9226502],"study_design_scores_gemma":[0.001210351,0.02324227,0.436339,0.01538329,0.02125275,0.003564087,0.003585388,0.2321429,0.111346,0.0676584,0.08259093,0.001684579],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1847833,0.1278953,0.6584609,0.001885534,0.0007137501,0.003107831,0.0102155,0.003477837,0.00946],"genre_scores_gemma":[0.5946142,0.02094457,0.3656338,0.0005874314,0.00052824,0.003279579,0.01132626,0.0008729079,0.002212978],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9570391,"threshold_uncertainty_score":0.2272014,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2148972377","doi":"10.1145/1571941.1572114","title":"Reciprocal rank fusion outperforms condorcet and individual rank learning methods","year":2009,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":568,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Reciprocal; Condorcet method; Rank (graph theory); Computer science; Simple (philosophy); Mean reciprocal rank; Fusion; Fuse (electrical); Artificial intelligence; Learning to rank; Machine learning; Mathematics; Ranking (information retrieval); Combinatorics; Engineering; Voting","authors":[{"name":"Gordon V. Cormack","is_ca":true},{"name":"Charles L. A. Clarke","is_ca":true},{"name":"Stefan Buettcher","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04029236246797672,"gpt":0.3261430280054581,"spread":0.2858506655374814,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01554707,0.002794731,0.002737776,0.005251543,0.002288471,0.003396012,0.002762933,0.002555219,0.006495238],"category_scores_gemma":[0.02099973,0.0004563766,0.002099348,0.004145848,0.001244684,0.00568732,0.002756121,0.003364278,0.005738601],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00199599,"about_ca_system_score_gemma":0.002802499,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008063242,"about_ca_topic_score_gemma":0.01253845,"domain_scores_codex":[0.9846497,0.00431429,0.000765527,0.002311881,0.007043764,0.0009148035],"domain_scores_gemma":[0.9879893,0.004027053,0.0006910707,0.003832174,0.003033789,0.0004265897],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00131588,0.0006315913,0.005160831,0.0005821103,0.0008779478,0.0000958363,0.0002095626,0.04700416,0.009096009,0.00690101,0.05460455,0.8735205],"study_design_scores_gemma":[0.0003429592,0.001844854,0.006793503,0.0001706639,0.0007020358,0.0008692479,0.0003470441,0.8564595,0.05348266,0.03150867,0.04719779,0.0002810071],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1458109,0.01680315,0.7205831,0.002949137,0.001442351,0.0008806008,0.00472032,0.05194733,0.05486314],"genre_scores_gemma":[0.5817881,0.002358673,0.3791952,0.0006820528,0.0007266329,0.0002913376,0.009000432,0.002226807,0.02373068],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01554707,"threshold_uncertainty_score":0.08222175,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4211256381","doi":"10.1109/access.2022.3151048","title":"MLCM: Multi-Label Confusion Matrix","year":2022,"lang":"en","type":"article","venue":"IEEE Access","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":409,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Toronto Metropolitan University; Vector Institute; McMaster University","funders":"Ministère de la Défense Nationale","keywords":"Computer science; Confusion; Confusion matrix; Artificial intelligence; Classifier (UML); Class (philosophy); Ambiguity; Machine learning; Programming language","authors":[{"name":"Mohammadreza Heydarian","is_ca":true},{"name":"Thomas E. Doyle","is_ca":true},{"name":"Reza Samavi","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08270770067299012,"gpt":0.3675705466220109,"spread":0.2848628459490208,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005264724,0.003701651,0.001887485,0.008003796,0.001862052,0.004771274,0.005331772,0.003175834,0.03626817],"category_scores_gemma":[0.02744752,0.00111726,0.002018406,0.004376899,0.001346187,0.006020645,0.005036145,0.004282397,0.02732536],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002237642,"about_ca_system_score_gemma":0.004852526,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007191714,"about_ca_topic_score_gemma":0.008582709,"domain_scores_codex":[0.9889236,0.003254606,0.00096604,0.001922103,0.004395531,0.0005382273],"domain_scores_gemma":[0.9861109,0.005012678,0.001227253,0.002500374,0.004664953,0.0004837859],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007385141,0.0003433214,0.001644938,0.001581808,0.0001941073,0.0004220669,0.0003862878,0.03430916,0.011061,0.03098334,0.2615465,0.6567889],"study_design_scores_gemma":[0.0001903997,0.0003516891,0.002654561,0.0004550941,0.00008576733,0.0008977198,0.0003039681,0.6371062,0.04776569,0.1204492,0.1892549,0.0004847844],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002471395,0.0007961969,0.9293271,0.0008850894,0.0005481102,0.0007746046,0.01029852,0.04942818,0.00547084],"genre_scores_gemma":[0.06476128,0.0006537208,0.8915112,0.001008067,0.0004546756,0.002335115,0.024719,0.003436018,0.0111208],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03626817,"threshold_uncertainty_score":0.1213291,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2142742813","doi":"","title":"Learning from multiple partially observed views - an application to multilingual text categorization","year":2009,"lang":"en","type":"preprint","venue":"NPARC","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":274,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Computer science; Artificial intelligence; Leverage (statistics); Categorization; Machine learning; Natural language processing; Generalization; Text categorization; Set (abstract data type); Training set; Supervised learning; Artificial neural network; Mathematics","authors":[{"name":"Massih Amini","is_ca":false},{"name":"Nicolas Usunier","is_ca":false},{"name":"Cyril Goutte","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0694166084151494,"gpt":0.2987423463949583,"spread":0.2293257379798089,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004723601,0.0008028169,0.001922352,0.001695621,0.0009681313,0.002045695,0.002167351,0.002009852,0.001276881],"category_scores_gemma":[0.02130318,0.0005661816,0.001641518,0.002484338,0.00141766,0.003935613,0.002788011,0.002574,0.0005233674],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00131182,"about_ca_system_score_gemma":0.0007067282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002197972,"about_ca_topic_score_gemma":0.002217289,"domain_scores_codex":[0.9968234,0.001548613,0.0001582067,0.000685739,0.000615775,0.0001682928],"domain_scores_gemma":[0.9819584,0.01208247,0.001298294,0.003036444,0.001250696,0.0003737239],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005738136,0.0002970904,0.01349335,0.0003846415,0.0005663804,0.001017653,0.00116333,0.3169044,0.00836513,0.05157679,0.006719757,0.5989377],"study_design_scores_gemma":[0.00002169243,0.00007696205,0.001324535,0.0000275074,0.00004751883,0.0002452997,0.000109719,0.9244691,0.003172743,0.06845555,0.002017877,0.00003153715],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02946595,0.0005994254,0.9681234,0.0005193119,0.00003547188,0.00004889567,0.00019099,0.0004421311,0.000574526],"genre_scores_gemma":[0.5818497,0.0006119032,0.4129444,0.0002925304,0.000296497,0.000207254,0.001802647,0.0001567289,0.001838237],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004723601,"threshold_uncertainty_score":0.02498102,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W167522204","doi":"","title":"Beyond TFIDF weighting for text categorization in the vector space model","year":2005,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":232,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université Laval","funders":"","keywords":"Weighting; tf–idf; Vector space model; Computer science; Artificial intelligence; Categorization; Leverage (statistics); Vocabulary; Feature selection; Support vector machine; Text categorization; Word (group theory); Feature vector; Natural language processing; Machine learning; Information retrieval; Pattern recognition (psychology); Mathematics; Term (time)","authors":[{"name":"Pascal Soucy","is_ca":false},{"name":"Guy W. Mineau","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02088681033409115,"gpt":0.2605201725384578,"spread":0.2396333622043666,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004120719,0.001110363,0.001611814,0.003532721,0.0007365138,0.002281934,0.00143613,0.001757777,0.003778939],"category_scores_gemma":[0.01265271,0.0002978698,0.001156874,0.005622689,0.0009035217,0.006875532,0.0007987497,0.001457312,0.002531298],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001252705,"about_ca_system_score_gemma":0.0009194022,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003041183,"about_ca_topic_score_gemma":0.002228802,"domain_scores_codex":[0.997409,0.001035793,0.0001810076,0.000384453,0.0008324778,0.0001573406],"domain_scores_gemma":[0.9953964,0.002480276,0.0002841828,0.0008226074,0.0009138953,0.0001025363],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000216534,0.0001735101,0.001849657,0.0003874978,0.0001277073,0.0001021339,0.000230868,0.06175534,0.008326645,0.09910979,0.008145613,0.8195748],"study_design_scores_gemma":[0.0000298901,0.0001337385,0.001124345,0.00006139974,0.00004522824,0.0002456597,0.00006912193,0.7415646,0.005202009,0.2354152,0.01603577,0.00007302973],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006951607,0.001867158,0.9883196,0.0003060808,0.0001719075,0.00008246818,0.0001237548,0.0004842509,0.001693144],"genre_scores_gemma":[0.2809679,0.003157531,0.702567,0.0003590392,0.000948815,0.0003813535,0.0009544413,0.0003191872,0.01034477],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004120719,"threshold_uncertainty_score":0.02179271,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2099126842","doi":"10.1109/icdm.2001.989592","title":"A simple KNN algorithm for text categorization","year":2002,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":225,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Interpretability; Artificial intelligence; Text categorization; Feature (linguistics); Context (archaeology); Vocabulary; Feature selection; Word (group theory); Categorization; Simple (philosophy); Class (philosophy); k-nearest neighbors algorithm; Process (computing); Machine learning; Natural language processing; Pattern recognition (psychology); Mathematics","authors":[{"name":"Pascal Soucy","is_ca":true},{"name":"Guy W. Mineau","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03016497877693925,"gpt":0.2537802900861292,"spread":0.22361531130919,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002183947,0.002020608,0.001527363,0.004818519,0.001920708,0.002100165,0.002703072,0.002684817,0.009330341],"category_scores_gemma":[0.007986929,0.0005954772,0.00146442,0.005167935,0.001001479,0.003672694,0.001564922,0.002167924,0.01295996],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001123079,"about_ca_system_score_gemma":0.002013379,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004765752,"about_ca_topic_score_gemma":0.006579669,"domain_scores_codex":[0.9964685,0.0005496865,0.0003592005,0.0009805036,0.001472677,0.0001694898],"domain_scores_gemma":[0.9981555,0.000533679,0.0001543539,0.0003339213,0.0007570608,0.00006551814],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001094679,0.0001143501,0.0006807648,0.000405163,0.0001243681,0.00008221358,0.0001017859,0.02014195,0.006105151,0.01052688,0.02219839,0.9394094],"study_design_scores_gemma":[0.0001593808,0.0003217575,0.002698133,0.0003399825,0.0001823433,0.001369898,0.0002358906,0.6179875,0.02085824,0.1748949,0.1806987,0.0002533497],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002073414,0.001005024,0.9890928,0.0002647378,0.0003586765,0.0004822347,0.0006074676,0.002709447,0.003406155],"genre_scores_gemma":[0.02206756,0.0007411701,0.9652344,0.0003310904,0.0002161739,0.0005842369,0.001518199,0.0002148558,0.009092454],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009330341,"threshold_uncertainty_score":0.03121305,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2089870669","doi":"10.1016/j.eswa.2011.09.160","title":"Comparison of term frequency and document frequency based feature selection metrics in text categorization","year":2011,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":152,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Regina","funders":"Faculty of Graduate Studies and Research, University of Alberta; Natural Sciences and Engineering Research Council of Canada; University of Regina","keywords":"Feature selection; Discriminative model; Computer science; Term (time); Text categorization; Categorization; Word lists by frequency; Feature (linguistics); Frequency; Artificial intelligence; Selection (genetic algorithm); Pattern recognition (psychology); tf–idf; Data mining; Mathematics; Statistics","authors":[{"name":"Nouman Azam","is_ca":true},{"name":"JingTao Yao","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03104747594243825,"gpt":0.2863871260576094,"spread":0.2553396501151711,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007369473,0.0007362022,0.001714788,0.008694715,0.0006183662,0.002287736,0.0008298984,0.001058735,0.0009872641],"category_scores_gemma":[0.02153089,0.000160184,0.0009252518,0.006065359,0.0003891039,0.003035834,0.0007324978,0.0007328248,0.0004117906],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009292995,"about_ca_system_score_gemma":0.001007283,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002635662,"about_ca_topic_score_gemma":0.003027074,"domain_scores_codex":[0.9952807,0.001381554,0.0006418823,0.000368515,0.002099152,0.0002282765],"domain_scores_gemma":[0.9680778,0.02402744,0.001336264,0.0008269638,0.0051425,0.0005889791],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003741232,0.0007666025,0.04127194,0.001211128,0.0008361935,0.0001132483,0.0003835254,0.01296392,0.01714732,0.002503321,0.006852606,0.9122091],"study_design_scores_gemma":[0.0006886059,0.006505141,0.2174399,0.0003248268,0.001431435,0.001343719,0.001517268,0.7219796,0.0291134,0.009271628,0.009999031,0.0003855258],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8060538,0.01900421,0.1651201,0.0008211363,0.0005060783,0.0003177764,0.002392485,0.001909372,0.003874871],"genre_scores_gemma":[0.8899647,0.002226198,0.1014041,0.00009452019,0.0003152272,0.000222533,0.003983019,0.0001574125,0.001632305],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008694715,"threshold_uncertainty_score":0.03897399,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2151849342","doi":"10.1109/hicss.2003.1174243","title":"Support vector machines for text categorization","year":2003,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":146,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Categorization; Text categorization; Sorting; Set (abstract data type); Support vector machine; Artificial intelligence; Vocabulary; Feature (linguistics); Natural language processing; Information retrieval; Process (computing); Feature vector","authors":[{"name":"Avik Basu","is_ca":true},{"name":"Clarence P. Walters","is_ca":true},{"name":"Michael Shepherd","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01982714791796406,"gpt":0.2626373654355585,"spread":0.2428102175175944,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00278022,0.001627902,0.002087573,0.003416009,0.0006423211,0.002614758,0.002107186,0.002061337,0.01038901],"category_scores_gemma":[0.01315567,0.0004352759,0.0009464442,0.005999781,0.0006826837,0.00311157,0.001307306,0.002698463,0.01102323],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008936721,"about_ca_system_score_gemma":0.001120002,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002385283,"about_ca_topic_score_gemma":0.001786532,"domain_scores_codex":[0.9962812,0.001365156,0.0003124421,0.0005207862,0.001375726,0.0001446001],"domain_scores_gemma":[0.9953393,0.002940573,0.0003091439,0.0004846941,0.0008327913,0.00009348035],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001579701,0.0001505094,0.001033974,0.001125599,0.0002155738,0.000158455,0.0001391394,0.04124416,0.001693278,0.04085144,0.04819452,0.8650354],"study_design_scores_gemma":[0.0001199484,0.0001562433,0.00228126,0.0005549477,0.00009367596,0.0003843851,0.0002207822,0.5490111,0.003298292,0.2931138,0.1506246,0.0001408404],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005207237,0.03559077,0.9315372,0.00307668,0.001297556,0.0007499054,0.003380095,0.007585241,0.0115754],"genre_scores_gemma":[0.1116541,0.02115308,0.8378341,0.0007252376,0.002204414,0.001767083,0.008144868,0.0004492958,0.016068],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01038901,"threshold_uncertainty_score":0.03475469,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1528447234","doi":"10.1145/563932.563930","title":"Classifying text documents by associating terms with text categories","year":2002,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":111,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Categorization; Computer science; Text categorization; Association rule learning; Classifier (UML); Artificial intelligence; Text mining; Natural language processing; Information retrieval; Machine learning; Data mining","authors":[{"name":"Osmar R. Zai͏̈ane","is_ca":true},{"name":"Maria-Luiza Antonie","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01833621184447505,"gpt":0.2287015926834513,"spread":0.2103653808389762,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001269478,0.0009086961,0.000881415,0.01298669,0.001018236,0.002209197,0.001007373,0.000993777,0.002425563],"category_scores_gemma":[0.006805511,0.0002043952,0.00101909,0.008686708,0.0006773715,0.003097951,0.0009225401,0.0008697693,0.002869238],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005075071,"about_ca_system_score_gemma":0.0008847766,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001470251,"about_ca_topic_score_gemma":0.001915609,"domain_scores_codex":[0.9986193,0.0002628947,0.0002286677,0.0002658584,0.0005233717,0.00009997378],"domain_scores_gemma":[0.9965927,0.001894401,0.0004019659,0.0002897149,0.0007147492,0.0001065816],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003378858,0.0005041799,0.02824619,0.00111262,0.000150192,0.0004711724,0.001097079,0.003096088,0.03532609,0.009688441,0.007814031,0.9121561],"study_design_scores_gemma":[0.0002879245,0.002605048,0.1350535,0.001730402,0.001836625,0.006552727,0.007628246,0.3856679,0.1045859,0.1871757,0.1663048,0.0005713453],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3063246,0.005923875,0.6530637,0.001555034,0.0009776915,0.002241283,0.006522781,0.004413128,0.01897789],"genre_scores_gemma":[0.2982845,0.00245758,0.6858177,0.0001860166,0.0004906188,0.0008433749,0.007085952,0.0001574359,0.004676811],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01298669,"threshold_uncertainty_score":0.008114278,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2012971775","doi":"10.1080/0952813x.2012.721010","title":"Naive Bayes text classifiers: a locally weighted learning approach","year":2012,"lang":"en","type":"article","venue":"Journal of Experimental & Theoretical Artificial Intelligence","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of New Brunswick","funders":"","keywords":"Naive Bayes classifier; Computer science; Artificial intelligence; Machine learning; Conditional independence; Bayes error rate; Bayesian programming; Bayes' theorem; Benchmark (surveying); Complement (music); Bayes classifier; Bayesian probability; Support vector machine; Bayes factor","authors":[{"name":"Liangxiao Jiang","is_ca":false},{"name":"Zhihua Cai","is_ca":false},{"name":"Harry Zhang","is_ca":true},{"name":"Dianhong Wang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03611338585351168,"gpt":0.3060859303784622,"spread":0.2699725445249506,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00700182,0.001987149,0.0031221,0.005481208,0.001453436,0.00289259,0.005024553,0.003028872,0.004048643],"category_scores_gemma":[0.02473951,0.0007887874,0.001564915,0.004585416,0.00134252,0.006020168,0.001807518,0.002959844,0.0033276],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001463237,"about_ca_system_score_gemma":0.001818757,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003620075,"about_ca_topic_score_gemma":0.004593103,"domain_scores_codex":[0.9906984,0.003562261,0.0007312726,0.001526256,0.003203726,0.0002781356],"domain_scores_gemma":[0.9893999,0.005684435,0.0007626645,0.001062913,0.002907779,0.0001823647],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000410105,0.0003534622,0.003629642,0.0006069218,0.0003706633,0.0002041394,0.0002766852,0.07446691,0.005205722,0.01796685,0.0173277,0.8791811],"study_design_scores_gemma":[0.00009218669,0.0002029081,0.0008692876,0.0001375246,0.000182655,0.000299779,0.0001155762,0.8990396,0.004123312,0.0853314,0.009533932,0.00007193008],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005819599,0.00103298,0.9888268,0.000411164,0.0001424671,0.0003434066,0.0002790961,0.001429065,0.001715448],"genre_scores_gemma":[0.1933293,0.001086913,0.7942524,0.0009873775,0.0008124894,0.00115595,0.002019601,0.0003061264,0.006049952],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00700182,"threshold_uncertainty_score":0.03702956,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2135420268","doi":"10.1109/grc.2007.40","title":"Na&amp;#x0EF;ve Bayes Text Classifier","year":2007,"lang":"en","type":"article","venue":"2007 IEEE International Conference on Granular Computing (GRC 2007)","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":98,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Acadia University","funders":"","keywords":"Naive Bayes classifier; Computer science; Bayes' theorem; Support vector machine; Artificial intelligence; Machine learning; Detector; Classifier (UML); Bayesian probability; Information retrieval; Data mining","authors":[{"name":"Haiyi Zhang","is_ca":true},{"name":"Di Li","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07050096873332103,"gpt":0.3294173998244482,"spread":0.2589164310911272,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001884984,0.0007581442,0.001385165,0.002727088,0.001107136,0.003427191,0.001705274,0.002022303,0.02024359],"category_scores_gemma":[0.007755195,0.0004214859,0.0006724125,0.001972695,0.0008148871,0.00337867,0.0008936563,0.001083396,0.01736278],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001034941,"about_ca_system_score_gemma":0.001144075,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003513016,"about_ca_topic_score_gemma":0.003807526,"domain_scores_codex":[0.9978637,0.0002779877,0.0001886377,0.0003707986,0.001138551,0.0001603781],"domain_scores_gemma":[0.9975968,0.0008626677,0.0001709323,0.0003415672,0.0009563059,0.00007176775],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002712108,0.000174459,0.004665375,0.0003580932,0.00005578523,0.0002440352,0.0001362161,0.008869835,0.01178922,0.04649443,0.03920837,0.887733],"study_design_scores_gemma":[0.00008832236,0.0001649514,0.00453983,0.0003481882,0.00009092948,0.001498584,0.0002313278,0.6756578,0.04065933,0.1106354,0.1659673,0.0001179334],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01555,0.001935342,0.9220532,0.001211342,0.0007366374,0.0004113109,0.00190241,0.007991184,0.04820859],"genre_scores_gemma":[0.2116734,0.001702125,0.7000576,0.0008921443,0.0005368983,0.0005492851,0.003275638,0.000566436,0.08074653],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02024359,"threshold_uncertainty_score":0.06772155,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2013023084","doi":"10.1007/s10994-013-5371-6","title":"Beam search algorithms for multilabel learning","year":2013,"lang":"en","type":"article","venue":"Machine Learning","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":90,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Inference; Beam search; Computer science; Machine learning; Artificial intelligence; Probabilistic logic; Classifier (UML); Task (project management); Range (aeronautics); Algorithm; Search algorithm","authors":[{"name":"Abhishek Kumar","is_ca":false},{"name":"Shankar Vembu","is_ca":true},{"name":"Aditya Krishna Menon","is_ca":false},{"name":"Charles Elkan","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03105901679861803,"gpt":0.289212411422877,"spread":0.2581533946242589,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006210132,0.001563785,0.003674974,0.003572394,0.002525618,0.002311798,0.004622538,0.003144595,0.01566523],"category_scores_gemma":[0.02268532,0.001454278,0.001805921,0.006047668,0.002337182,0.005827612,0.004292376,0.004821149,0.005715663],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001579334,"about_ca_system_score_gemma":0.002321513,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00429512,"about_ca_topic_score_gemma":0.007780662,"domain_scores_codex":[0.995754,0.002378638,0.0002037212,0.0006267076,0.0007644841,0.0002724942],"domain_scores_gemma":[0.9878536,0.008808124,0.0003592756,0.001569836,0.001154901,0.0002542759],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006126372,0.0003801314,0.001034229,0.0005225143,0.0002646075,0.00008997994,0.0003410903,0.1698957,0.002198189,0.2318424,0.04309098,0.5497276],"study_design_scores_gemma":[0.0001355964,0.00007144243,0.0001456021,0.00006759887,0.00004493077,0.00005318804,0.00005828809,0.5778378,0.001183539,0.4140254,0.006344605,0.00003192413],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001677271,0.0005739401,0.9951504,0.000209863,0.00005212282,0.00004868135,0.0001427063,0.0008877399,0.00125729],"genre_scores_gemma":[0.04675668,0.0006233525,0.9441584,0.0005588239,0.0001817332,0.0005935036,0.00127262,0.0007118732,0.005142992],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01566523,"threshold_uncertainty_score":0.05240542,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2997205428","doi":"10.1609/aaai.v34i04.5822","title":"Nonlinear Mixup: Out-Of-Manifold Data Augmentation for Text Classification","year":2020,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":81,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"National Research Council Canada","funders":"","keywords":"Nonlinear system; Mixing (physics); Manifold (fluid mechanics); Computer science; Regularization (linguistics); Scalar (mathematics); Interpolation (computer graphics); Benchmark (surveying); Mathematics; Artificial intelligence; Pattern recognition (psychology); Algorithm; Machine learning","authors":[{"name":"Hongyu Guo","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2956569575793979,"gpt":0.3630008410034393,"spread":0.06734388342404135,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002016166,0.00217636,0.001479946,0.00122536,0.0009773485,0.001370628,0.002649165,0.001988772,0.004744065],"category_scores_gemma":[0.00595257,0.0007946804,0.001484927,0.001318619,0.001597533,0.003501836,0.005103257,0.003577281,0.002813751],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000826434,"about_ca_system_score_gemma":0.0008967377,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001319419,"about_ca_topic_score_gemma":0.00271347,"domain_scores_codex":[0.9987906,0.0004382427,0.00006426145,0.0003676912,0.0002349033,0.0001043424],"domain_scores_gemma":[0.9982742,0.0006533727,0.0001353984,0.0005863233,0.0002353894,0.0001152211],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008731891,0.0005563645,0.004100112,0.0003903285,0.0002101019,0.0002788579,0.0006943284,0.1984309,0.02545085,0.02385407,0.01790032,0.7272605],"study_design_scores_gemma":[0.00002732424,0.0001363181,0.0002549285,0.00001975024,0.00001759155,0.00007585284,0.00005772798,0.9725211,0.009211336,0.01404434,0.003610401,0.00002343248],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03094624,0.0005147695,0.9602191,0.0003350704,0.0001056475,0.0001732421,0.000369863,0.005567543,0.001768477],"genre_scores_gemma":[0.4172451,0.0003564764,0.5690773,0.0007831632,0.0001746066,0.0007257066,0.002615784,0.0009361979,0.008085658],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004744065,"threshold_uncertainty_score":0.01587045,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3121217868","doi":"10.1016/j.is.2021.101718","title":"Multi-label legal document classification: A deep learning-based approach with label-attention and domain-specific pre-training","year":2021,"lang":"en","type":"article","venue":"Information Systems","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Computer science; Artificial intelligence; Multi-label classification; Domain (mathematical analysis); Training (meteorology); Machine learning; Training set","authors":[{"name":"Dezhao Song","is_ca":false},{"name":"Andrew Vold","is_ca":false},{"name":"Kanika Madan","is_ca":true},{"name":"Frank Schilder","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03582749930198761,"gpt":0.2545681606306928,"spread":0.2187406613287052,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001647781,0.001595032,0.001547157,0.00306448,0.001296409,0.002041968,0.003697452,0.002844192,0.005518426],"category_scores_gemma":[0.003075048,0.0006066195,0.00134016,0.002655275,0.0006670802,0.003144622,0.002437753,0.004677767,0.003903836],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001993077,"about_ca_system_score_gemma":0.002934932,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0187812,"about_ca_topic_score_gemma":0.03737109,"domain_scores_codex":[0.9987261,0.0001782531,0.00008935053,0.0004162437,0.000323773,0.000266426],"domain_scores_gemma":[0.9975689,0.0007699335,0.0001841106,0.0004338741,0.000845594,0.0001974988],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003630729,0.001044306,0.002197383,0.0001791673,0.0001045106,0.0001294315,0.0001194567,0.01744452,0.01337623,0.003505846,0.03213159,0.9294044],"study_design_scores_gemma":[0.0000318825,0.00007231294,0.0007225749,0.00003942103,0.00005807085,0.00007363729,0.0000784345,0.976003,0.01143343,0.006745298,0.004716859,0.00002511691],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0714076,0.002294987,0.893616,0.002249743,0.0008290017,0.00051509,0.002856842,0.01746476,0.008766009],"genre_scores_gemma":[0.391906,0.0009645445,0.5704145,0.001121866,0.0005521409,0.0003537508,0.01052983,0.0006387031,0.02351882],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0187812,"threshold_uncertainty_score":0.03734374,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2000918452","doi":"10.1145/1281192.1281260","title":"A concept-based model for enhancing text categorization","year":2007,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Natural language processing; Categorization; Sentence; Artificial intelligence; Phrase; Term (time); Weighting; Graph; Information retrieval; Theoretical computer science","authors":[{"name":"Shady Shehata","is_ca":true},{"name":"Fakhri Karray","is_ca":true},{"name":"Mohamed S. Kamel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02597749126124376,"gpt":0.281605414501903,"spread":0.2556279232406592,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002313172,0.001244986,0.0006617145,0.003740012,0.0007282894,0.001664525,0.002314668,0.001192433,0.003666173],"category_scores_gemma":[0.007057735,0.0003522958,0.001326572,0.00341783,0.0007239598,0.004648527,0.001177576,0.001542793,0.00168244],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001681884,"about_ca_system_score_gemma":0.002022671,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006619163,"about_ca_topic_score_gemma":0.005640667,"domain_scores_codex":[0.9982322,0.0004046697,0.0001125968,0.0004109778,0.0007659274,0.00007378394],"domain_scores_gemma":[0.9977648,0.001174707,0.0001220963,0.0001925168,0.0006892347,0.00005650875],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003021224,0.0004592792,0.002476509,0.0007628257,0.0001967264,0.0003268193,0.0009649023,0.069766,0.01919918,0.108221,0.01414671,0.7831779],"study_design_scores_gemma":[0.00004167233,0.0001371727,0.001150161,0.00008617753,0.00009389679,0.0003372068,0.0001487144,0.9024665,0.005554552,0.06749175,0.02243443,0.00005765586],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006018725,0.0003177314,0.9897192,0.0002867509,0.00008181383,0.0004144969,0.0003569523,0.001070405,0.001733999],"genre_scores_gemma":[0.1037561,0.0004685099,0.8906437,0.0002651718,0.00009284219,0.001050457,0.0009986327,0.0001004563,0.002624123],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006619163,"threshold_uncertainty_score":0.01316124,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4386266758","doi":"10.1109/access.2023.3309697","title":"MCNN-LSTM: Combining CNN and LSTM to Classify Multi-Class Text in Imbalanced News Data","year":2023,"lang":"en","type":"article","venue":"IEEE Access","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":64,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Artificial intelligence; Convolutional neural network; Machine learning; Margin (machine learning); Deep learning; Class (philosophy); Feature (linguistics); Sentence; Document classification; Categorization; Natural language processing","authors":[{"name":"Khan Md. Hasib","is_ca":false},{"name":"Sami Azam","is_ca":false},{"name":"Asif Karim","is_ca":false},{"name":"Ahmed Al Marouf","is_ca":true},{"name":"F. M. Javed Mehedi Shamrat","is_ca":false},{"name":"Sidratul Montaha","is_ca":false},{"name":"Kheng Cher Yeo","is_ca":false},{"name":"Mirjam Jonkman","is_ca":false},{"name":"Reda Alhajj","is_ca":true},{"name":"Jon Rokne","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1552556872135712,"gpt":0.3857469680310411,"spread":0.2304912808174699,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005728624,0.001274465,0.0005917131,0.001069756,0.0003620354,0.0006900115,0.001028789,0.0008176141,0.00148173],"category_scores_gemma":[0.001255963,0.0003298592,0.0005010486,0.001212834,0.0003031387,0.001882767,0.0006585082,0.001017143,0.0007589893],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007505972,"about_ca_system_score_gemma":0.0006361061,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008392334,"about_ca_topic_score_gemma":0.01175682,"domain_scores_codex":[0.9997186,0.00002555256,0.00001685047,0.0001081793,0.00006792093,0.00006295028],"domain_scores_gemma":[0.9996903,0.00007944544,0.00005351731,0.00004054013,0.0001071217,0.00002909697],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004410241,0.0003168959,0.00510209,0.0001972261,0.0002143979,0.0003702825,0.0001487787,0.09680742,0.03299803,0.001647909,0.01922414,0.8425318],"study_design_scores_gemma":[0.00001432008,0.00008695576,0.001269538,0.00001915265,0.00003999696,0.0000801103,0.0000367758,0.9851351,0.009297798,0.002080008,0.00192492,0.00001525201],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2601503,0.005218315,0.7049825,0.001532168,0.00151438,0.0002502821,0.002126133,0.01495206,0.009273904],"genre_scores_gemma":[0.8432761,0.001177529,0.1420313,0.0007501463,0.000440033,0.0001741954,0.003341642,0.000291201,0.008517942],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008392334,"threshold_uncertainty_score":0.01668698,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2018764947","doi":"10.1016/j.ipm.2011.11.001","title":"A three-phase method for patent classification","year":2012,"lang":"en","type":"article","venue":"Information Processing & Management","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"McGill University","keywords":"Phase (matter); Computer science; Information retrieval; Chemistry","authors":[{"name":"Yen‐Liang Chen","is_ca":false},{"name":"Yuan-Che Chang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07881797612441159,"gpt":0.3411085603720877,"spread":0.2622905842476761,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003104604,0.0010154,0.001260554,0.004556916,0.001715253,0.00244083,0.002346829,0.001784221,0.01024655],"category_scores_gemma":[0.007058891,0.0005350533,0.001466261,0.003911751,0.000556438,0.002568915,0.001671259,0.001551221,0.006802034],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008270312,"about_ca_system_score_gemma":0.004447795,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006260076,"about_ca_topic_score_gemma":0.008352148,"domain_scores_codex":[0.9965357,0.0007104812,0.0004261113,0.0005461478,0.001531743,0.0002497808],"domain_scores_gemma":[0.994586,0.001856291,0.000189959,0.0006553967,0.002551396,0.0001609746],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002854557,0.0002872501,0.001127434,0.0001259203,0.0000664552,0.00004699145,0.00008551697,0.003128846,0.00967293,0.006452172,0.008813701,0.9699073],"study_design_scores_gemma":[0.0004032726,0.0007558193,0.006789749,0.0001002441,0.000333603,0.001095983,0.0002838929,0.8277965,0.05756605,0.04124877,0.06340288,0.0002232357],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004905673,0.0002250301,0.9894459,0.0001936315,0.0001352947,0.0005765103,0.0004108954,0.002383457,0.001723648],"genre_scores_gemma":[0.04043419,0.0001603298,0.9501134,0.000148764,0.0001197797,0.0006085243,0.001616445,0.0001588808,0.006639783],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01024655,"threshold_uncertainty_score":0.03427809,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4394987794","doi":"10.1016/j.datak.2024.102306","title":"Effective text classification using BERT, MTM LSTM, and DT","year":2024,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Artificial Intelligence in Medicine (Canada); University of Calgary","funders":"","keywords":"Security token; Computer science; Artificial intelligence; Encoder; Binary classification; Transformer; Recall; Deep learning; Long short term memory; Natural language processing; Artificial neural network; Machine learning; Recurrent neural network; Support vector machine","authors":[{"name":"Saman Jamshidi","is_ca":false},{"name":"Mahin Mohammadi","is_ca":false},{"name":"Saeed Bagheri","is_ca":false},{"name":"Hamid Esmaeili Najafabadi","is_ca":true},{"name":"Alireza Rezvanian","is_ca":true},{"name":"Mehdi Gheisari","is_ca":true},{"name":"Mustafa Ghaderzadeh","is_ca":true},{"name":"Amir Shahab Shahabi","is_ca":false},{"name":"Zongda Wu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04931219236959951,"gpt":0.3094819476933262,"spread":0.2601697553237267,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006206296,0.0008233437,0.0007244822,0.001811987,0.0006673686,0.001094075,0.001024357,0.00129214,0.005158818],"category_scores_gemma":[0.002448098,0.0002291545,0.0006030879,0.001813399,0.0003055074,0.002656406,0.0007310713,0.001283318,0.003326725],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009054172,"about_ca_system_score_gemma":0.001572894,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007583996,"about_ca_topic_score_gemma":0.00974217,"domain_scores_codex":[0.9995552,0.00007035082,0.00004527409,0.0001395463,0.0001268222,0.0000628143],"domain_scores_gemma":[0.9990849,0.0003254322,0.00007557225,0.0001244204,0.0003266997,0.00006282092],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002427495,0.0001341855,0.0007072202,0.0001461462,0.00004037334,0.00009943236,0.00004500004,0.02009593,0.02194989,0.004845452,0.0138783,0.9378154],"study_design_scores_gemma":[0.0000180675,0.00007866747,0.000664972,0.00002412015,0.00003873727,0.0001180951,0.00003939693,0.9720643,0.01342392,0.008044233,0.005468566,0.00001688484],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04909409,0.00225709,0.9276817,0.001263351,0.001143188,0.0001620683,0.001718087,0.009320917,0.007359538],"genre_scores_gemma":[0.4653713,0.001210194,0.5054748,0.0007186317,0.0007777089,0.0002237309,0.004140392,0.0004296021,0.02165369],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007583996,"threshold_uncertainty_score":0.01725793,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2108345839","doi":"10.1177/0165551508092257","title":"A hidden Markov model-based text classification of medical documents","year":2008,"lang":"en","type":"article","venue":"Journal of Information Science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"","keywords":"Hidden Markov model; Computer science; Categorization; Artificial intelligence; Markov model; Subject (documents); Natural language processing; Machine learning; Markov chain; World Wide Web","authors":[{"name":"Kwan Yi","is_ca":false},{"name":"Jamshid Beheshti","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02964063379902278,"gpt":0.2983874154971723,"spread":0.2687467816981495,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001795575,0.0005354066,0.0008464137,0.001724407,0.0004713062,0.001165124,0.001056932,0.001068148,0.002199278],"category_scores_gemma":[0.006967328,0.0002602147,0.000878119,0.001216137,0.0003178653,0.001773588,0.0004139399,0.001056769,0.00179488],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001160139,"about_ca_system_score_gemma":0.001376586,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01483398,"about_ca_topic_score_gemma":0.01069629,"domain_scores_codex":[0.9988139,0.0003275331,0.000102336,0.0003121675,0.0003546186,0.00008958808],"domain_scores_gemma":[0.9956735,0.002906199,0.0001960138,0.0002544122,0.000839536,0.0001304593],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001744755,0.0009196413,0.01590413,0.0006540463,0.0002946329,0.0003648116,0.0003703933,0.1134657,0.02445914,0.003963219,0.01293637,0.8249232],"study_design_scores_gemma":[0.00004598331,0.0001929999,0.003438154,0.00003316361,0.00006103074,0.000148706,0.00004238893,0.9867159,0.006181827,0.001623211,0.001488917,0.00002769376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1538474,0.001187647,0.8259616,0.001206991,0.0004110569,0.0008215813,0.003179192,0.009982266,0.003402288],"genre_scores_gemma":[0.6417464,0.0006013308,0.3455542,0.0003325453,0.000145371,0.0004956438,0.005753186,0.0001547435,0.005216562],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01483398,"threshold_uncertainty_score":0.0294953,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2756654724","doi":"10.18653/v1/d17-1202","title":"Shortest-Path Graph Kernels for Document Similarity","year":2017,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Graph; Similarity measure; Shortest path problem; Kernel (algebra); Similarity (geometry); Macro; Data mining; Artificial intelligence; Pattern recognition (psychology); Theoretical computer science; Mathematics; Combinatorics","authors":[{"name":"Giannis Nikolentzos","is_ca":false},{"name":"Polykarpos Meladianos","is_ca":false},{"name":"François Rousseau","is_ca":true},{"name":"Yannis Stavrakas","is_ca":false},{"name":"Michalis Vazirgiannis","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04094903064209897,"gpt":0.3136976083326138,"spread":0.2727485776905149,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001366543,0.0007786215,0.00118146,0.004747808,0.0004898114,0.00159605,0.001300497,0.001106022,0.001535813],"category_scores_gemma":[0.009130389,0.0002246146,0.0007870893,0.00552367,0.0008999269,0.005089116,0.00129743,0.001277796,0.001031114],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001147037,"about_ca_system_score_gemma":0.0007558967,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001821679,"about_ca_topic_score_gemma":0.001186729,"domain_scores_codex":[0.9973346,0.0006555648,0.0002291485,0.0005704549,0.001082519,0.0001278019],"domain_scores_gemma":[0.9950593,0.002129908,0.0006528549,0.0009255527,0.001073095,0.0001593847],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006155827,0.0003732977,0.005890562,0.0007464065,0.0003382791,0.0002450389,0.0003933897,0.14022,0.03008263,0.07771613,0.006129683,0.737249],"study_design_scores_gemma":[0.00002306225,0.000225315,0.003655941,0.00003942317,0.00006190482,0.0005177412,0.0001285729,0.9097321,0.01122952,0.06648511,0.007823278,0.0000779541],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03325514,0.00143483,0.9622416,0.0001094651,0.00007016034,0.0000899673,0.0002652803,0.001194364,0.001339145],"genre_scores_gemma":[0.6163772,0.001070422,0.3785362,0.00008578769,0.000141903,0.0001967743,0.001258654,0.0002962133,0.002036894],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004747808,"threshold_uncertainty_score":0.008322358,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2263083451","doi":"","title":"Bi-parameter space partition for cost-sensitive SVM","year":2015,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Western University","funders":"","keywords":"Support vector machine; Partition (number theory); Hyperparameter optimization; Regularization (linguistics); Algorithm; Invariant (physics); Parameter space; Piecewise; Grid; Model selection; Computer science; Mathematical optimization; Mathematics; Generalization; Artificial intelligence; Statistics; Combinatorics; Mathematical analysis; Geometry","authors":[{"name":"Bin Gu","is_ca":true},{"name":"Victor S. Sheng","is_ca":false},{"name":"Shuo Li","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08266643554195872,"gpt":0.3065639875173696,"spread":0.2238975519754109,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002433142,0.00170067,0.001502979,0.001658347,0.0007431163,0.001662149,0.001747847,0.001576649,0.00192277],"category_scores_gemma":[0.006497066,0.0007112158,0.001061916,0.001090579,0.0009967738,0.002335669,0.002011064,0.002323167,0.0005625646],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009880724,"about_ca_system_score_gemma":0.00123136,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002546074,"about_ca_topic_score_gemma":0.001655765,"domain_scores_codex":[0.9986399,0.0005361622,0.0001042288,0.0002818957,0.0003347537,0.0001029805],"domain_scores_gemma":[0.9982988,0.0008677303,0.0001914942,0.0002311937,0.0003419152,0.0000688691],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002151103,0.0001246891,0.001491215,0.0001622102,0.0001128897,0.000111496,0.0002098415,0.7665601,0.009225172,0.01901533,0.002890868,0.1998811],"study_design_scores_gemma":[0.000004679604,0.00002464035,0.00009550995,0.000005111677,0.000005986098,0.00001703293,0.00001186451,0.9943824,0.0008508241,0.004280955,0.0003150363,0.000005857097],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01153224,0.0002023806,0.9871985,0.00008827046,0.00001470847,0.00005099065,0.00002986227,0.0004167906,0.0004661508],"genre_scores_gemma":[0.5100931,0.0003564492,0.4857171,0.0002167201,0.00005992119,0.0005107263,0.0006421108,0.0004091685,0.001994727],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.002546074,"threshold_uncertainty_score":0.01286781,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2740468505","doi":"10.24963/ijcai.2017/443","title":"Incomplete Label Distribution Learning","year":2017,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Novelis (Canada)","funders":"National Natural Science Foundation of China","keywords":"Computer science; Artificial intelligence; Relevance (law); Minification; Exploit; Annotation; Machine learning; Algorithm","authors":[{"name":"Miao Xu","is_ca":false},{"name":"Zhi‐Hua Zhou","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04191246432389344,"gpt":0.2960959009144437,"spread":0.2541834365905503,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003570825,0.00109261,0.002021607,0.001735992,0.0009825404,0.001820692,0.003499474,0.00241275,0.003205969],"category_scores_gemma":[0.01015867,0.0006240053,0.001022294,0.00212739,0.001877733,0.004076401,0.002608004,0.003169478,0.001491362],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001614664,"about_ca_system_score_gemma":0.002323806,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002216732,"about_ca_topic_score_gemma":0.002862428,"domain_scores_codex":[0.9970386,0.001042386,0.0001172871,0.0009068191,0.0006857255,0.0002092085],"domain_scores_gemma":[0.9935463,0.003152334,0.0005918266,0.001282242,0.001142381,0.000284897],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003556864,0.0003494298,0.003835458,0.0004008995,0.0001585423,0.0002602027,0.0003219795,0.2927642,0.004993747,0.09621532,0.02543266,0.5749118],"study_design_scores_gemma":[0.00002822028,0.00003614262,0.0002480063,0.00002025133,0.00001510192,0.00005630066,0.0000302994,0.9301171,0.001925438,0.06409354,0.003413541,0.00001608581],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005369783,0.0001665705,0.9923984,0.0003345247,0.00003864965,0.00004604569,0.0001584072,0.000630335,0.0008572649],"genre_scores_gemma":[0.2703261,0.0005615058,0.7173388,0.0007642113,0.0004121135,0.0003790739,0.003023707,0.0003508667,0.006843699],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003570825,"threshold_uncertainty_score":0.01888454,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1507852567","doi":"10.1007/11425274_17","title":"Evaluation of Two Systems on Multi-class Multi-label Document Classification","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Document classification; Classifier (UML); Search engine indexing; Artificial intelligence; Class (philosophy); Multi-label classification; Pattern recognition (psychology); Data mining; Information retrieval; Natural language processing; Machine learning","authors":[{"name":"Xiao Luo","is_ca":true},{"name":"A. Nur Zincir‐Heywood","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09346704776640828,"gpt":0.3400993522796007,"spread":0.2466323045131924,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00945974,0.002469026,0.002711565,0.0038274,0.001766494,0.003998599,0.003864193,0.005224612,0.005424119],"category_scores_gemma":[0.02041075,0.0007269921,0.001524887,0.002612311,0.0008746097,0.00419083,0.00268649,0.001717889,0.003680371],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002657626,"about_ca_system_score_gemma":0.002294772,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02392591,"about_ca_topic_score_gemma":0.02083808,"domain_scores_codex":[0.9917483,0.002373959,0.0007552522,0.00195937,0.002524631,0.000638555],"domain_scores_gemma":[0.9754066,0.01329193,0.0006326172,0.002281641,0.007264553,0.001122644],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.01623368,0.004835855,0.01340099,0.001821449,0.00204644,0.0004792889,0.0005113481,0.03543305,0.0298293,0.001151974,0.02570026,0.8685564],"study_design_scores_gemma":[0.002325866,0.005412925,0.02732792,0.0001753684,0.002014998,0.0007698672,0.0008694829,0.8907587,0.05857226,0.001802532,0.009689447,0.0002805505],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7420449,0.01170741,0.194922,0.002428705,0.003425441,0.003116972,0.005228391,0.02288345,0.01424271],"genre_scores_gemma":[0.7330969,0.00176722,0.2293224,0.0008671067,0.0007351464,0.001082996,0.01642063,0.0008146125,0.01589291],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02392591,"threshold_uncertainty_score":0.05002844,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2998605053","doi":"10.1609/aaai.v34i04.6132","title":"Partial Label Learning with Batch Label Correction","year":2020,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Carleton University","funders":"China Scholarship Council; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Consistency (knowledge bases); Artificial intelligence; Machine learning; Set (abstract data type); Multi-label classification; Stability (learning theory)","authors":[{"name":"Yan Yan","is_ca":false},{"name":"Yuhong Guo","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1001122140207756,"gpt":0.2912905940668809,"spread":0.1911783800461053,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002525376,0.001652648,0.002014742,0.0008515705,0.00118752,0.001495839,0.00551633,0.001959894,0.002440396],"category_scores_gemma":[0.007624056,0.0006431843,0.001006114,0.001155231,0.001733022,0.004541608,0.002925672,0.003626861,0.001615802],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001328421,"about_ca_system_score_gemma":0.002621869,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006911833,"about_ca_topic_score_gemma":0.01102684,"domain_scores_codex":[0.997988,0.0004552938,0.00007506773,0.000810423,0.000489582,0.0001816345],"domain_scores_gemma":[0.9933345,0.002095874,0.0005122162,0.002260242,0.001512064,0.0002850067],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009523591,0.0004455413,0.00566482,0.0002525418,0.0001302139,0.0001942332,0.0003775407,0.1539407,0.02074547,0.008697346,0.01377662,0.7948226],"study_design_scores_gemma":[0.00004542218,0.0001114778,0.0003838966,0.00001268378,0.00002621466,0.00008414788,0.00004535309,0.9788193,0.009954377,0.008324006,0.002165573,0.00002772593],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03213263,0.000401154,0.958631,0.0002970817,0.0001441479,0.0001518114,0.0001946038,0.006571786,0.001475763],"genre_scores_gemma":[0.4253453,0.0002055575,0.5644311,0.000659043,0.0001830609,0.0003243151,0.001501788,0.0006497108,0.006700087],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006911833,"threshold_uncertainty_score":0.01374322,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1714192071","doi":"","title":"Categorical proportional difference: a feature selection method for text categorization","year":2008,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Regina","funders":"","keywords":"Artificial intelligence; Computer science; Feature selection; Categorical variable; Word (group theory); Task (project management); Naive Bayes classifier; Mutual information; Natural language processing; Feature (linguistics); Selection (genetic algorithm); Categorization; Measure (data warehouse); Support vector machine; Pattern recognition (psychology); Machine learning; Data mining; Mathematics","authors":[{"name":"Mondelle Simeon","is_ca":true},{"name":"Robert J. Hilderman","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02953612645523518,"gpt":0.2917414232733742,"spread":0.262205296818139,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004528372,0.001081919,0.001374531,0.005162566,0.0006889248,0.0009572151,0.001712007,0.0008905767,0.002825053],"category_scores_gemma":[0.009728521,0.0002733685,0.001318201,0.003916801,0.0006571818,0.001412556,0.001131039,0.001136066,0.001186595],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006465095,"about_ca_system_score_gemma":0.0008573253,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001420061,"about_ca_topic_score_gemma":0.001300865,"domain_scores_codex":[0.9960334,0.001160751,0.0003344697,0.0008092428,0.001481434,0.0001806744],"domain_scores_gemma":[0.9956702,0.002294959,0.000312341,0.0004862781,0.001129531,0.0001066317],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007381495,0.0002630784,0.005129711,0.0002919036,0.0002448856,0.0001514952,0.0002043282,0.009801404,0.02092265,0.00302503,0.008724689,0.9505026],"study_design_scores_gemma":[0.0004030557,0.001231861,0.02220967,0.00008361142,0.0002502277,0.001556286,0.0002846791,0.8563497,0.05742226,0.02786576,0.0320473,0.0002954968],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02636944,0.0006221407,0.9667667,0.0001954142,0.0002027513,0.0005156865,0.0008954974,0.003328197,0.001104319],"genre_scores_gemma":[0.2976147,0.0002333767,0.6957486,0.0002840815,0.0002198112,0.001108064,0.002049441,0.0002480662,0.002493838],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005162566,"threshold_uncertainty_score":0.02394861,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2137320444","doi":"","title":"Linear Text Segmentation Using Affinity Propagation","year":2011,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Affinity propagation; Segmentation; Computer science; Cluster analysis; Pairwise comparison; Graph; Image segmentation; Set (abstract data type); Algorithm; Artificial intelligence; Pattern recognition (psychology); Theoretical computer science; Correlation clustering; Canopy clustering algorithm","authors":[{"name":"Anna Kazantseva","is_ca":true},{"name":"Stan Śzpakowicz","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.103551508256095,"gpt":0.2887300140038151,"spread":0.1851785057477202,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00111943,0.001633384,0.001444413,0.003622498,0.001500208,0.002048339,0.003039748,0.002228359,0.00836444],"category_scores_gemma":[0.004958447,0.0009543838,0.001521515,0.004258817,0.001325095,0.003250159,0.002025628,0.001861477,0.00717466],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001492435,"about_ca_system_score_gemma":0.001583955,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01062943,"about_ca_topic_score_gemma":0.01316296,"domain_scores_codex":[0.9979253,0.000318261,0.0001263371,0.0007417654,0.0007129268,0.0001753507],"domain_scores_gemma":[0.9970855,0.001236278,0.0002489887,0.0005333815,0.0007886857,0.0001072231],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003331077,0.0001508206,0.0009342075,0.0002968357,0.0001433133,0.0001748877,0.0004973929,0.0889039,0.02862635,0.01397757,0.0182354,0.8477263],"study_design_scores_gemma":[0.00005170206,0.00008048434,0.0004710501,0.00002198891,0.00005113773,0.0001968594,0.00009662606,0.9342556,0.02117094,0.02978793,0.01376988,0.00004584184],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00349435,0.0001898734,0.990109,0.0001030884,0.00005055442,0.0000911537,0.0001246891,0.004366174,0.001471281],"genre_scores_gemma":[0.0773012,0.0002548117,0.9092598,0.000247545,0.0001397383,0.0002767697,0.001239862,0.001171316,0.01010893],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01062943,"threshold_uncertainty_score":0.02798182,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2131912994","doi":"10.1109/icdew.2007.4401066","title":"Document Representation and Dimension Reduction for Text Clustering","year":2007,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"","keywords":"Cluster analysis; Dimensionality reduction; Computer science; Document clustering; Artificial intelligence; Pattern recognition (psychology); Representation (politics); Context (archaeology); Benchmark (surveying); Word (group theory); Dimension (graph theory); Natural language processing; Mathematics","authors":[{"name":"Mahdi Shafiei","is_ca":true},{"name":"Singer Wang","is_ca":true},{"name":"Roger Zhang","is_ca":true},{"name":"Evangelos Milios","is_ca":true},{"name":"Bin Tang","is_ca":true},{"name":"Jane E. Tougas","is_ca":true},{"name":"Raymond J. Spiteri","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02785592813353458,"gpt":0.3078267137122061,"spread":0.2799707855786715,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00209821,0.0009500743,0.001277553,0.0043619,0.0009551603,0.001797058,0.001006586,0.0007763233,0.002335153],"category_scores_gemma":[0.009199312,0.0002691926,0.001228747,0.006258083,0.0004582229,0.001467797,0.0008825315,0.001028969,0.002324112],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009154171,"about_ca_system_score_gemma":0.001340195,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002438364,"about_ca_topic_score_gemma":0.002241694,"domain_scores_codex":[0.9976168,0.0009675795,0.0002387008,0.0003712818,0.0007036518,0.0001019902],"domain_scores_gemma":[0.9972253,0.001115284,0.0002260035,0.0005962054,0.0007885131,0.00004875025],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001942384,0.0001513128,0.001232644,0.0004834591,0.0001570737,0.00008418165,0.0003218447,0.03769486,0.01159173,0.01660562,0.01646832,0.9150147],"study_design_scores_gemma":[0.0001077144,0.0002868542,0.005146157,0.0001462612,0.0002045409,0.0006515923,0.0004957599,0.857314,0.02451032,0.06671607,0.04425116,0.0001696028],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0158168,0.001844474,0.975069,0.0005475485,0.000178619,0.0003578377,0.001513622,0.00266266,0.002009549],"genre_scores_gemma":[0.07137626,0.0009033469,0.9219764,0.00008674904,0.0001702403,0.0006139782,0.003092804,0.0001503311,0.001629881],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0043619,"threshold_uncertainty_score":0.01109648,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1586413166","doi":"10.3115/1118935.1118941","title":"Text classification in Asian languages without word segmentation","year":2003,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Text segmentation; Artificial intelligence; Natural language processing; Language model; n-gram; Word (group theory); Segmentation; Feature (linguistics); Character (mathematics); Key (lock); Word lists by frequency; Simple (philosophy); Linguistics; Mathematics","authors":[{"name":"Fuchun Peng","is_ca":true},{"name":"Dale Schuurmans","is_ca":true},{"name":"Shaojun Wang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02359207191929053,"gpt":0.2946768134136925,"spread":0.2710847414944019,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000644347,0.0008699421,0.000695827,0.002129838,0.0007928103,0.001081589,0.0006515451,0.0006578367,0.002544974],"category_scores_gemma":[0.001999168,0.0001981988,0.0008201497,0.002126963,0.0003478452,0.001772937,0.000721372,0.0005608591,0.002446208],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003494449,"about_ca_system_score_gemma":0.0008471343,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002543521,"about_ca_topic_score_gemma":0.006312219,"domain_scores_codex":[0.999467,0.0001098192,0.00006783788,0.0001375969,0.0001582447,0.00005963389],"domain_scores_gemma":[0.9986501,0.0003737377,0.000169541,0.0002475239,0.000476374,0.00008267027],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003661027,0.0003388096,0.01259771,0.0003420381,0.0001695308,0.0003694402,0.0006331161,0.01162648,0.1224418,0.004964353,0.007673885,0.8384768],"study_design_scores_gemma":[0.00007017403,0.0005493285,0.02092507,0.0000634908,0.0002906357,0.001016431,0.001027859,0.775539,0.1587322,0.01542877,0.02621811,0.0001388488],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1754274,0.0006635594,0.8086016,0.0004115114,0.0001761542,0.0003496536,0.001255684,0.005490781,0.007623725],"genre_scores_gemma":[0.5110043,0.0003151391,0.476091,0.0002240894,0.0001557871,0.0002749219,0.002820223,0.0003064341,0.008808168],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002544974,"threshold_uncertainty_score":0.008513749,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2079238049","doi":"10.1007/s13042-013-0152-x","title":"Bayesian Citation-KNN with distance weighting","year":2013,"lang":"en","type":"article","venue":"International Journal of Machine Learning and Cybernetics","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of New Brunswick","funders":"National Natural Science Foundation of China","keywords":"Computer science; Artificial intelligence; Bayesian probability; Machine learning; Benchmark (surveying); k-nearest neighbors algorithm; Weighting; Computational intelligence; Field (mathematics); Naive Bayes classifier; Pattern recognition (psychology); Overhead (engineering); Supervised learning; Data mining; Mathematics; Support vector machine; Artificial neural network","authors":[{"name":"Liangxiao Jiang","is_ca":false},{"name":"Zhihua Cai","is_ca":false},{"name":"Dianhong Wang","is_ca":false},{"name":"Harry Zhang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.005946413716180115,"gpt":0.2347103549350663,"spread":0.2287639412188862,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007079375,0.001222094,0.003286397,0.01078067,0.001871004,0.003345824,0.005201856,0.005288942,0.007809098],"category_scores_gemma":[0.03516333,0.0008688503,0.001918924,0.01208584,0.001119566,0.006248127,0.003070057,0.003027754,0.00398914],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00195137,"about_ca_system_score_gemma":0.003143903,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01448209,"about_ca_topic_score_gemma":0.01614871,"domain_scores_codex":[0.9939712,0.002076567,0.0006319299,0.001339177,0.001574108,0.0004070445],"domain_scores_gemma":[0.9861239,0.007171228,0.0009575634,0.00189491,0.003445163,0.0004071667],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006468562,0.0006759435,0.01103259,0.0005580534,0.0005859004,0.0001887582,0.0001650888,0.2827952,0.0009357083,0.05138329,0.02764233,0.6233903],"study_design_scores_gemma":[0.00004981102,0.00004626513,0.001070658,0.00005983579,0.0000932097,0.00008915005,0.00003005039,0.9174433,0.0004660681,0.07663409,0.003979842,0.00003776752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04829486,0.004075808,0.9306629,0.001787022,0.0007945629,0.0003968849,0.002866356,0.002244766,0.008876915],"genre_scores_gemma":[0.5781449,0.002474668,0.369456,0.0007148728,0.001390764,0.000809592,0.00906573,0.0004178722,0.03752558],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01448209,"threshold_uncertainty_score":0.03743976,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3008638314","doi":"10.1016/j.ijar.2020.02.003","title":"Label distribution learning: A local collaborative mechanism","year":2020,"lang":"en","type":"article","venue":"International Journal of Approximate Reasoning","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"National Natural Science Foundation of China","keywords":"Robustness (evolution); Ambiguity; Computer science; Artificial intelligence; Machine learning; Feature learning; Representation (politics); Pattern recognition (psychology)","authors":[{"name":"Suping Xu","is_ca":true},{"name":"Hengrong Ju","is_ca":true},{"name":"Lin Shang","is_ca":false},{"name":"Witold Pedrycz","is_ca":true},{"name":"Xibei Yang","is_ca":false},{"name":"Chun Li","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0166668925147492,"gpt":0.2669003154745818,"spread":0.2502334229598325,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01123617,0.001230187,0.002622547,0.003092006,0.003619496,0.004811798,0.009424263,0.004842258,0.011719],"category_scores_gemma":[0.02775939,0.001037653,0.001937623,0.004701944,0.002437192,0.009820425,0.01242648,0.00408987,0.003638513],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001918971,"about_ca_system_score_gemma":0.002964992,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004482917,"about_ca_topic_score_gemma":0.007033485,"domain_scores_codex":[0.991616,0.002773653,0.0004938073,0.001977005,0.002586063,0.0005535596],"domain_scores_gemma":[0.9773299,0.008463179,0.001055975,0.009253772,0.002993935,0.0009031407],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001585966,0.001535808,0.004174915,0.0002805904,0.0003520279,0.0002177737,0.0008456961,0.06008149,0.01283623,0.07498785,0.01741615,0.8256854],"study_design_scores_gemma":[0.0002357703,0.0001871297,0.0006243092,0.00003528003,0.0001463094,0.0001997961,0.0001639663,0.8668132,0.01424122,0.1095221,0.007769173,0.00006170964],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006738469,0.0001444155,0.989838,0.000360709,0.00005443379,0.00009991998,0.00008652839,0.001248868,0.001428705],"genre_scores_gemma":[0.3171343,0.0002127832,0.6692164,0.0005201558,0.0003503738,0.0004141781,0.0006568087,0.0004442349,0.01105076],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.011719,"threshold_uncertainty_score":0.05942327,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2237248385","doi":"10.1007/978-3-642-33460-3_48","title":"Learning and Inference in Probabilistic Classifier Chains with Beam Search","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Inference; Classifier (UML); Probabilistic logic; Artificial intelligence; Machine learning; Beam search; Binary classification; Flexibility (engineering); sort; Algorithm; Search algorithm; Mathematics; Support vector machine; Information retrieval","authors":[{"name":"Abhishek Kumar","is_ca":false},{"name":"Shankar Vembu","is_ca":true},{"name":"Aditya Krishna Menon","is_ca":false},{"name":"Charles Elkan","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02746902263249762,"gpt":0.2691181450713802,"spread":0.2416491224388826,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01499515,0.001550222,0.004875774,0.003343147,0.002454499,0.004008562,0.006254529,0.004940788,0.01039473],"category_scores_gemma":[0.05001458,0.004856129,0.003809658,0.005949952,0.004654101,0.01072453,0.006081265,0.007489037,0.00356664],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002497848,"about_ca_system_score_gemma":0.003160084,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0123673,"about_ca_topic_score_gemma":0.01449561,"domain_scores_codex":[0.9914796,0.004816509,0.0005948519,0.001397478,0.00119837,0.0005131058],"domain_scores_gemma":[0.9270695,0.06483648,0.001437644,0.003847884,0.002199425,0.0006089656],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006678761,0.0002949316,0.00279558,0.000347475,0.0003458931,0.0002025665,0.0005573147,0.6065794,0.0009928923,0.1474454,0.009013185,0.2307575],"study_design_scores_gemma":[0.00003661502,0.00002178148,0.00005324666,0.00002401835,0.00002027238,0.00002073807,0.0000149532,0.9011373,0.0002624444,0.0979967,0.0004013357,0.00001064174],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004281921,0.0003010268,0.9937896,0.0002778428,0.00002392229,0.00005509502,0.0001086492,0.0005631029,0.0005989092],"genre_scores_gemma":[0.1273734,0.00052597,0.8648536,0.0005216624,0.0002529192,0.0005752939,0.001388452,0.000395199,0.004113509],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01499515,"threshold_uncertainty_score":0.07930285,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1983349993","doi":"10.1016/j.neunet.2006.12.005","title":"The learning vector quantization algorithm applied to automatic text classification tasks","year":2007,"lang":"en","type":"article","venue":"Neural Networks","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Vedecká Grantová Agentúra MŠVVaŠ SR a SAV; McGill University","keywords":"Learning vector quantization; Computer science; Artificial intelligence; Classifier (UML); Artificial neural network; Word-sense disambiguation; Categorization; Linde–Buzo–Gray algorithm; Text categorization; Machine learning; Word (group theory); Natural language processing; Pattern recognition (psychology); Vector quantization; Mathematics; WordNet","authors":[{"name":"María Teresa Martín Valdivia","is_ca":false},{"name":"Luís Alfonso Ureña López","is_ca":false},{"name":"Manuel Garcı́a-Vega","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02001477650850889,"gpt":0.2704345014242594,"spread":0.2504197249157505,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002721036,0.0005407652,0.001211904,0.001318652,0.0007354119,0.001184923,0.001267029,0.001076578,0.003176144],"category_scores_gemma":[0.007320526,0.0003133346,0.0005132307,0.002607695,0.0005690123,0.001807026,0.000810207,0.001651829,0.001467781],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000841377,"about_ca_system_score_gemma":0.001793717,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007613204,"about_ca_topic_score_gemma":0.005919899,"domain_scores_codex":[0.9980572,0.000636417,0.0002461133,0.0002992599,0.0006674416,0.00009345454],"domain_scores_gemma":[0.9974632,0.0009684917,0.00009944834,0.0003438172,0.001068327,0.00005667725],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000143103,0.00008235258,0.0005915553,0.0001671414,0.00004837215,0.00002872237,0.00007223318,0.0184757,0.007407563,0.0117642,0.008830606,0.9523886],"study_design_scores_gemma":[0.00008841324,0.0002148443,0.001639084,0.00008788572,0.00005864924,0.0001887763,0.00007013982,0.9209679,0.01896534,0.04273682,0.01493035,0.00005173739],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01149018,0.002299721,0.9807321,0.0005180922,0.0006244482,0.000197028,0.000351956,0.00184414,0.001942343],"genre_scores_gemma":[0.1694007,0.001636605,0.8197133,0.0003808842,0.000343949,0.0004247065,0.001219782,0.0002052846,0.006674839],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007613204,"threshold_uncertainty_score":0.01513779,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2214660060","doi":"10.1609/aaai.v25i1.7895","title":"Adaptive Large Margin Training for Multilabel Classification","year":2011,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Margin (machine learning); Computer science; Exploit; Scalability; Categorization; Machine learning; Generalization; Classifier (UML); Artificial intelligence; Training set; Data mining; Pattern recognition (psychology); Mathematics","authors":[{"name":"Yuhong Guo","is_ca":false},{"name":"Dale Schuurmans","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3253993417235344,"gpt":0.3314708791762336,"spread":0.006071537452699172,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005187968,0.001480617,0.001647686,0.001034465,0.00104788,0.001259314,0.003427844,0.002181152,0.003645554],"category_scores_gemma":[0.01501662,0.0006411471,0.00107262,0.001342876,0.001819822,0.004583553,0.00408839,0.004162673,0.002069129],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001232822,"about_ca_system_score_gemma":0.001139694,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001468991,"about_ca_topic_score_gemma":0.002302049,"domain_scores_codex":[0.9964374,0.001756515,0.0001520188,0.0007965035,0.0006197404,0.0002377923],"domain_scores_gemma":[0.9933971,0.00371977,0.0005121395,0.001377979,0.0007355163,0.0002574342],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007034817,0.0005782457,0.002688319,0.0002570968,0.0001271561,0.0001716279,0.0003139319,0.5211056,0.008927869,0.03894596,0.01574311,0.4104376],"study_design_scores_gemma":[0.00001375392,0.00004035873,0.0001548197,0.000009455988,0.000005848043,0.00002773537,0.00002077077,0.9672228,0.001837114,0.02966992,0.0009883652,0.000009138029],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01130823,0.0002137818,0.9859719,0.0002919786,0.00003284857,0.00004585216,0.00009055045,0.001139111,0.0009057312],"genre_scores_gemma":[0.4172128,0.0002935059,0.5731238,0.0006985712,0.0002330022,0.000485912,0.001588226,0.0007897108,0.00557449],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005187968,"threshold_uncertainty_score":0.02743697,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2609579487","doi":"10.1007/s11280-017-0460-2","title":"Collaborative text categorization via exploiting sparse coefficients","year":2017,"lang":"en","type":"article","venue":"World Wide Web","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Computer science; Sparse approximation; Pattern recognition (psychology); Categorization; Artificial intelligence; Classifier (UML); Sparse matrix; Representation (politics); Information retrieval; Data mining","authors":[{"name":"Lina Yao","is_ca":false},{"name":"Quan Z. Sheng","is_ca":false},{"name":"Xianzhi Wang","is_ca":false},{"name":"Shengrui Wang","is_ca":true},{"name":"Xue Li","is_ca":false},{"name":"Sen Wang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01904446666712901,"gpt":0.2666869600097453,"spread":0.2476424933426163,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001039275,0.0007729222,0.001407873,0.004075421,0.0008184388,0.001451958,0.001301072,0.001222697,0.00219616],"category_scores_gemma":[0.004806999,0.0003958253,0.000966198,0.00352702,0.0004805933,0.002811892,0.00176868,0.001170128,0.001696236],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004194613,"about_ca_system_score_gemma":0.000844189,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00340367,"about_ca_topic_score_gemma":0.005892926,"domain_scores_codex":[0.9988062,0.0002391671,0.00007466954,0.0002542822,0.0004995948,0.000126082],"domain_scores_gemma":[0.996905,0.001509002,0.0002799237,0.00048147,0.0006754043,0.0001492499],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005525615,0.0006493763,0.003017458,0.0001998581,0.0001809689,0.0001968423,0.0002231165,0.04360488,0.0467491,0.01006373,0.01474541,0.8798166],"study_design_scores_gemma":[0.00003424434,0.00008110131,0.0007365484,0.00001174867,0.00005348908,0.0001106288,0.00007466516,0.9734312,0.008777551,0.01401389,0.002656884,0.0000179988],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04322628,0.0005607653,0.9517264,0.0003677576,0.0001691433,0.0001090631,0.0003622453,0.001467447,0.002010853],"genre_scores_gemma":[0.5059493,0.0006078337,0.4804784,0.0003417723,0.0004967255,0.0002492589,0.003068652,0.000205427,0.008602547],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004075421,"threshold_uncertainty_score":0.007346869,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1960913922","doi":"10.1007/3-540-44886-1_41","title":"Feature Selection Strategies for Text Categorization","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université Laval","funders":"","keywords":"Feature selection; Categorization; Computer science; Selection (genetic algorithm); Text categorization; Feature (linguistics); Rank (graph theory); Artificial intelligence; Set (abstract data type); Filter (signal processing); Function (biology); Representation (politics); Machine learning; Data mining; Pattern recognition (psychology); Mathematics","authors":[{"name":"Pascal Soucy","is_ca":true},{"name":"Guy W. Mineau","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01843416811592578,"gpt":0.2540111248480112,"spread":0.2355769567320854,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001643236,0.0009862849,0.001322434,0.002467728,0.0007472325,0.001281381,0.001686222,0.0008831742,0.004477978],"category_scores_gemma":[0.00348859,0.0003949352,0.0009094627,0.002632889,0.0003853787,0.001821872,0.0008740416,0.0009872472,0.002495321],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003834056,"about_ca_system_score_gemma":0.0006250866,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001972272,"about_ca_topic_score_gemma":0.002656563,"domain_scores_codex":[0.9992136,0.0002337535,0.00009495334,0.0001552819,0.0002283412,0.00007403511],"domain_scores_gemma":[0.9982754,0.001000469,0.00006025261,0.0002014834,0.0004116729,0.0000507922],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001587807,0.0001144039,0.0004653648,0.0001160476,0.00005772994,0.00007733833,0.00007926731,0.005447129,0.009766662,0.003072864,0.009604596,0.9710397],"study_design_scores_gemma":[0.0002283097,0.0004439759,0.004419132,0.0001057735,0.0002889644,0.000740144,0.0003157842,0.8501797,0.03715748,0.08321512,0.022817,0.00008865974],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01752741,0.002686456,0.9744018,0.0002294571,0.0001670631,0.0001775198,0.0004362455,0.002417955,0.00195606],"genre_scores_gemma":[0.1952155,0.001547232,0.7884749,0.0002622606,0.0003760878,0.0005928451,0.003174931,0.0004473863,0.009908844],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004477978,"threshold_uncertainty_score":0.01498032,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2077512673","doi":"10.1214/07-ba209","title":"Improving classification when a class hierarchy is available using a hierarchy-based prior","year":2007,"lang":"en","type":"article","venue":"Bayesian Analysis","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Hierarchy; Multinomial logistic regression; Class hierarchy; Computer science; Class (philosophy); Machine learning; Artificial intelligence; Multinomial distribution; Bayesian probability; Tree (set theory); Data mining; Mathematics; Statistics","authors":[{"name":"Radford M. Neal","is_ca":true},{"name":"Babak Shahbaba","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03222386111863169,"gpt":0.2757000272695841,"spread":0.2434761661509524,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005549299,0.001144436,0.001703351,0.003321329,0.001146953,0.00198854,0.002189211,0.00191499,0.003505332],"category_scores_gemma":[0.02292203,0.0008976844,0.001612213,0.002541037,0.001185026,0.00606089,0.002414818,0.004192213,0.002173964],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001581483,"about_ca_system_score_gemma":0.001946552,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01353026,"about_ca_topic_score_gemma":0.01987591,"domain_scores_codex":[0.9958107,0.001294628,0.0001961766,0.0008479599,0.001547048,0.0003034296],"domain_scores_gemma":[0.9855419,0.009237544,0.0008422449,0.002480401,0.001523305,0.0003745367],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004801848,0.0005965664,0.01262573,0.0002473934,0.0002444366,0.0001127593,0.0008138675,0.1058982,0.01087823,0.02547402,0.01493039,0.8276982],"study_design_scores_gemma":[0.00004302853,0.00007484404,0.002860883,0.00004952322,0.00005930132,0.00008897795,0.0000800624,0.9521767,0.003221023,0.03763674,0.003667615,0.00004143557],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02513021,0.0004353309,0.9697622,0.0006038241,0.00004494441,0.00007052,0.0002442473,0.001701898,0.002006726],"genre_scores_gemma":[0.3425012,0.0004245429,0.6512994,0.0004625507,0.0002209419,0.0001943356,0.001284802,0.0003733022,0.003238867],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01353026,"threshold_uncertainty_score":0.02934784,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W160585330","doi":"10.1007/978-3-642-33486-3_23","title":"Semi-supervised Multi-label Classification","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Artificial intelligence; Semi-supervised learning; Machine learning; Subspace topology; Margin (machine learning); Supervised learning; Classifier (UML); Exploit; Pattern recognition (psychology); Labeled data; Unsupervised learning; Artificial neural network","authors":[{"name":"Yuhong Guo","is_ca":false},{"name":"Dale Schuurmans","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0571843465909377,"gpt":0.2826659220794802,"spread":0.2254815754885425,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044231,0.00175181,0.002719058,0.002332288,0.001323941,0.002696101,0.003743363,0.002447656,0.004359585],"category_scores_gemma":[0.006894827,0.0008225886,0.00212209,0.002159458,0.0009893351,0.003016459,0.002956023,0.002110884,0.007113028],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006716262,"about_ca_system_score_gemma":0.001580486,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001283434,"about_ca_topic_score_gemma":0.003816212,"domain_scores_codex":[0.9942274,0.001842287,0.0004857717,0.001655103,0.001477461,0.0003118902],"domain_scores_gemma":[0.9892265,0.004197069,0.0006902558,0.002591437,0.003007493,0.0002873624],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004486255,0.0004147531,0.002049817,0.0005809622,0.0003233039,0.0001675284,0.0001927657,0.02205377,0.01707948,0.003654638,0.02599479,0.9270396],"study_design_scores_gemma":[0.00003869928,0.0001980121,0.001686094,0.0001346856,0.0001469869,0.0005329966,0.0001708945,0.9432368,0.02123776,0.02092277,0.01162567,0.00006866507],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01209864,0.0008525261,0.9781055,0.0002802525,0.0002705631,0.0002333641,0.0008924859,0.004107382,0.003159391],"genre_scores_gemma":[0.2026156,0.000550981,0.7694378,0.0004720025,0.0004325461,0.0006722781,0.009060482,0.0008256786,0.01593255],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0044231,"threshold_uncertainty_score":0.02339184,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3194282644","doi":"10.1186/s10033-021-00598-9","title":"ML-ANet: A Transfer Learning Approach Using Adaptation Network for Multi-label Image Classification in Autonomous Driving","year":2021,"lang":"en","type":"article","venue":"Chinese Journal of Mechanical Engineering","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Shenzhen Fundamental Research and Discipline Layout project; National Natural Science Foundation of China","keywords":"Reproducing kernel Hilbert space; Artificial intelligence; Computer science; Pattern recognition (psychology); Transfer of learning; Kernel (algebra); Embedding; Image (mathematics); Domain adaptation; Feature (linguistics); Adaptation (eye); Matching (statistics); Mathematics; Hilbert space; Statistics","authors":[{"name":"Guofa Li","is_ca":true},{"name":"Zefeng Ji","is_ca":false},{"name":"Yunlong Chang","is_ca":false},{"name":"Shen Li","is_ca":false},{"name":"Xingda Qu","is_ca":false},{"name":"Dongpu Cao","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04901093335217047,"gpt":0.2809918205667821,"spread":0.2319808872146116,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001254118,0.0009482982,0.0008369309,0.0007640582,0.0006629467,0.0007498375,0.001945328,0.001364151,0.001921058],"category_scores_gemma":[0.002083472,0.0003494388,0.0008050484,0.0006920446,0.0005773189,0.001831617,0.001269519,0.001673986,0.0005808527],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008498527,"about_ca_system_score_gemma":0.0007202801,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006492347,"about_ca_topic_score_gemma":0.005209309,"domain_scores_codex":[0.9994025,0.000143646,0.00003044562,0.0002215067,0.0001135657,0.00008823598],"domain_scores_gemma":[0.99907,0.0002972409,0.00007608369,0.0001026203,0.0003912803,0.00006287752],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003860451,0.0004456008,0.003641354,0.0001086331,0.0001496897,0.0001967784,0.0001458166,0.4790505,0.009766749,0.003750085,0.00636657,0.4959923],"study_design_scores_gemma":[0.000003484836,0.00002387547,0.0001724084,0.000002470977,0.000006015333,0.00001065799,0.00001128598,0.9976904,0.0008307159,0.0009780895,0.0002657713,0.000004803156],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1129305,0.0008180676,0.8789112,0.0005131982,0.0003351322,0.0001300124,0.0001814916,0.002588996,0.003591424],"genre_scores_gemma":[0.8813351,0.0002777517,0.1082997,0.0004224272,0.0001562686,0.0001543016,0.0006104896,0.0001134353,0.008630419],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006492347,"threshold_uncertainty_score":0.01290911,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2340990333","doi":"10.1177/0165551515625030","title":"The impact of indexing approaches on Arabic text classification","year":2016,"lang":"en","type":"article","venue":"Journal of Information Science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Classifier (UML); Search engine indexing; Arabic; Artificial intelligence; Naive Bayes classifier; Natural language processing; Pattern recognition (psychology); Word (group theory); Mathematics; Support vector machine","authors":[{"name":"Amer Al‐Badarneh","is_ca":false},{"name":"Emad Al‐Shawakfa","is_ca":false},{"name":"Basel Bani‐Ismail","is_ca":false},{"name":"Khaleel Al-Rababah","is_ca":true},{"name":"Safwan Shatnawi","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06173573844272501,"gpt":0.3028807703285067,"spread":0.2411450318857817,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01002092,0.001663661,0.001440257,0.004119969,0.001964013,0.004069255,0.001203585,0.001331207,0.001262582],"category_scores_gemma":[0.04268224,0.000314755,0.001078802,0.004588072,0.001007941,0.005105977,0.001754612,0.001416223,0.002071179],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001282804,"about_ca_system_score_gemma":0.001486847,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006092732,"about_ca_topic_score_gemma":0.00605629,"domain_scores_codex":[0.9861135,0.005568518,0.002086437,0.001580959,0.004045274,0.0006053484],"domain_scores_gemma":[0.9625334,0.02616159,0.001832206,0.002572795,0.006363461,0.0005366582],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002294645,0.0009363249,0.03883403,0.0009303804,0.0003842525,0.0002083059,0.0008870443,0.0110615,0.02534007,0.0009733319,0.004488742,0.9136614],"study_design_scores_gemma":[0.0004333887,0.00944893,0.1683647,0.001047567,0.002156492,0.003922848,0.009303888,0.5481142,0.2160703,0.01091584,0.02953406,0.0006877108],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9017443,0.01549091,0.06092926,0.001202547,0.0008277833,0.0006419016,0.001166588,0.0026184,0.01537821],"genre_scores_gemma":[0.895936,0.003308946,0.09432859,0.0002772021,0.0003515175,0.0001883304,0.002883717,0.0002385144,0.002487407],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01002092,"threshold_uncertainty_score":0.05299634,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4390873538","doi":"10.1109/iccv51070.2023.00166","title":"Parametric Information Maximization for Generalized Category Discovery","year":2023,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Institut National de la Recherche Scientifique; Thales (Canada)","funders":"","keywords":"Parameterized complexity; Maximization; Computer science; Parametric statistics; Mutual information; Class (philosophy); Code (set theory); Data mining; Nonparametric statistics; Artificial intelligence; Theoretical computer science; Machine learning; Algorithm; Mathematics; Mathematical optimization; Statistics; Set (abstract data type); Programming language","authors":[{"name":"Florent Chiaroni","is_ca":true},{"name":"José Dolz","is_ca":false},{"name":"Ziko Imtiaz Masud","is_ca":true},{"name":"Amar Mitiche","is_ca":true},{"name":"Ismail Ben Ayed","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02811367513141904,"gpt":0.2630281952364658,"spread":0.2349145201050468,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01189839,0.001856275,0.003075664,0.002884216,0.001089771,0.003645578,0.005688806,0.003336981,0.004230587],"category_scores_gemma":[0.03312018,0.001262746,0.002533157,0.004707407,0.003135193,0.006028085,0.005237724,0.005038107,0.002099916],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00296908,"about_ca_system_score_gemma":0.002941845,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003086949,"about_ca_topic_score_gemma":0.00342867,"domain_scores_codex":[0.9919515,0.004251707,0.0003421092,0.001576055,0.001415825,0.0004629196],"domain_scores_gemma":[0.9855042,0.01012719,0.0008954348,0.001898468,0.001211121,0.0003636008],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002566085,0.0002309524,0.002829272,0.0005498456,0.0003816794,0.0001975453,0.0003999193,0.538397,0.001962686,0.2177768,0.01940701,0.2176107],"study_design_scores_gemma":[0.00001843342,0.00003409231,0.0002236857,0.0000326857,0.00001812543,0.00006959794,0.00002278561,0.7738621,0.0005827347,0.2224585,0.002654487,0.00002297278],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003461767,0.0003873372,0.9937484,0.0005441083,0.00002016228,0.00006915834,0.0003251996,0.0004021929,0.001041603],"genre_scores_gemma":[0.23461,0.0009553937,0.7523061,0.001204806,0.0002848875,0.001129234,0.003388817,0.0006607611,0.005460047],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01189839,"threshold_uncertainty_score":0.0629254,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1997305228","doi":"10.1109/nlpke.2010.5587767","title":"Automatic classification of documents by formality","year":2010,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Formality; Computer science; Naive Bayes classifier; Artificial intelligence; Feature selection; Classifier (UML); Support vector machine; Machine learning; Decision tree; Task (project management); Style (visual arts); Data mining; Natural language processing; Engineering","authors":[{"name":"Fadi Abu Sheikha","is_ca":true},{"name":"Diana Inkpen","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01347406047579485,"gpt":0.2704840205753174,"spread":0.2570099600995225,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001812396,0.0007710324,0.0008019537,0.009146565,0.0007853853,0.002232893,0.0006996752,0.0005198816,0.001936466],"category_scores_gemma":[0.012998,0.0002437104,0.0006863637,0.004923573,0.0004716464,0.002609891,0.0007840706,0.0009791016,0.001435208],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007588942,"about_ca_system_score_gemma":0.0008199032,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00172188,"about_ca_topic_score_gemma":0.001926642,"domain_scores_codex":[0.9977398,0.0004214641,0.0002947435,0.0003813098,0.0009721749,0.000190486],"domain_scores_gemma":[0.9874471,0.005786377,0.001770901,0.001154043,0.003453638,0.000387961],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005422566,0.0002951619,0.06395314,0.000605014,0.0001124339,0.0002532456,0.0006561862,0.005414502,0.03024001,0.005104483,0.01131719,0.8815064],"study_design_scores_gemma":[0.0002541945,0.0007862133,0.2143134,0.0006087339,0.0003826074,0.00268039,0.00220722,0.5464178,0.1013696,0.05300361,0.07757604,0.00040012],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6312392,0.00434199,0.3294421,0.001078757,0.0005233919,0.0007694658,0.009952767,0.009130212,0.01352211],"genre_scores_gemma":[0.7617556,0.001061202,0.2217862,0.00009763128,0.0003179365,0.000231658,0.01096443,0.0002525846,0.003532722],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009146565,"threshold_uncertainty_score":0.009585023,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3086863200","doi":"10.1002/ett.3977","title":"Digital Hadith authentication: Recent advances, open challenges, and future directions","year":2020,"lang":"en","type":"article","venue":"Transactions on Emerging Telecommunications Technologies","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Northern British Columbia","funders":"Deanship of Scientific Research, King Saud University","keywords":"Authentication (law); Islam; Computer science; Field (mathematics); Legislation; Task (project management); Computer security; Political science; Law; History; Engineering; Systems engineering; Mathematics; Archaeology","authors":[{"name":"Saqib Hakak","is_ca":true},{"name":"Amirrudin Kamsin","is_ca":false},{"name":"Wazir Zada Khan","is_ca":false},{"name":"Abubakar Zakari","is_ca":false},{"name":"Muhammad Imran","is_ca":false},{"name":"Khadher Ahmad","is_ca":false},{"name":"Gulshan Amin Gilkar","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04322460988914023,"gpt":0.2839163424881219,"spread":0.2406917325989817,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005372778,0.0006707694,0.0007717561,0.003080757,0.001147621,0.004511557,0.002162163,0.002759904,0.006782006],"category_scores_gemma":[0.007678505,0.0004605613,0.0006310738,0.003512379,0.002400831,0.01458319,0.002106685,0.002245443,0.003117142],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00111242,"about_ca_system_score_gemma":0.002060018,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001332424,"about_ca_topic_score_gemma":0.001107822,"domain_scores_codex":[0.9971126,0.0008201762,0.0002566462,0.0004998791,0.001009944,0.0003006499],"domain_scores_gemma":[0.9872351,0.006483432,0.0006840806,0.0009500142,0.004083202,0.000564147],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001354934,0.0001911379,0.002829431,0.003352088,0.00003932155,0.0001801402,0.0006313021,0.00150674,0.001918076,0.03671008,0.01736579,0.9351403],"study_design_scores_gemma":[0.00004657364,0.0007891143,0.006075722,0.005669374,0.0001875583,0.002908421,0.01103059,0.038492,0.008148319,0.09557705,0.8308148,0.0002604396],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.01635995,0.8769118,0.04598714,0.02674552,0.002211248,0.0001707032,0.0001792689,0.00049709,0.03093731],"genre_scores_gemma":[0.2296444,0.7005264,0.04919813,0.004258942,0.004148817,0.0001509761,0.000701284,0.00006680587,0.01130412],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.006782006,"threshold_uncertainty_score":0.02841431,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2943818116","doi":"10.5539/mas.v13n5p88","title":"Arabic Text Classification: A Review","year":2019,"lang":"en","type":"review","venue":"Modern Applied Science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Computer science; Arabic; Weighting; Artificial intelligence; Classifier (UML); Decision tree; Natural language processing; Naive Bayes classifier; Support vector machine; Set (abstract data type); Data mining; Information retrieval; Machine learning; Linguistics","authors":[{"name":"Adel Hamdan Mohammad","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.126081744370099,"gpt":0.3611282266457068,"spread":0.2350464822756078,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001426114,0.0009556832,0.001126213,0.009377225,0.000739789,0.002564368,0.001668914,0.001059965,0.007461914],"category_scores_gemma":[0.005755672,0.0003646212,0.0008152401,0.008854106,0.0008086629,0.00460763,0.0006940776,0.001089602,0.00603976],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008805445,"about_ca_system_score_gemma":0.001479277,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002790734,"about_ca_topic_score_gemma":0.002627939,"domain_scores_codex":[0.9991779,0.000105197,0.0001544485,0.0001604414,0.0003585609,0.00004341445],"domain_scores_gemma":[0.9952214,0.002460669,0.0003445951,0.0001242837,0.001731392,0.0001175471],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00003341746,0.00004159714,0.0005514341,0.004809873,0.00003213989,0.00006754653,0.00007119864,0.0002016249,0.0003820654,0.001282959,0.04130991,0.9512163],"study_design_scores_gemma":[0.0000131705,0.00009912159,0.004391262,0.00550977,0.0001429601,0.001417093,0.000363984,0.001654003,0.001754179,0.003779084,0.9808235,0.00005189379],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.001589495,0.9751058,0.006538305,0.003323648,0.001753965,0.0001058186,0.0005355269,0.0002541476,0.01079331],"genre_scores_gemma":[0.009669708,0.9709648,0.008230756,0.001090458,0.002586032,0.000103955,0.001287316,0.00006329266,0.006003683],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.009377225,"threshold_uncertainty_score":0.02496254,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2143765837","doi":"10.1109/icmla.2005.47","title":"Multi-label Associative Classification of Medical Documents from MEDLINE","year":2006,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Associative property; Information retrieval; Document classification; Point (geometry); Multi-label classification; Data mining; Artificial intelligence; Machine learning","authors":[{"name":"Rafał Rak","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true},{"name":"Marek Reformat","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03854488586979353,"gpt":0.3163884499854844,"spread":0.2778435641156908,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001236381,0.0003293765,0.0004671548,0.006363043,0.0005265431,0.0009897943,0.0006285281,0.0005671094,0.002132724],"category_scores_gemma":[0.007685007,0.00007727937,0.0003338996,0.00363881,0.0002525558,0.001286478,0.000579509,0.0003491918,0.001017249],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003272706,"about_ca_system_score_gemma":0.0007733282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009198012,"about_ca_topic_score_gemma":0.001823555,"domain_scores_codex":[0.9991336,0.0002026873,0.0001597575,0.0001060981,0.0003417465,0.00005613939],"domain_scores_gemma":[0.9916831,0.004651818,0.0009289371,0.0008308067,0.001681178,0.0002241047],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009517444,0.0004298469,0.02613099,0.0004722439,0.00007018509,0.0003583646,0.0003450216,0.004652339,0.02982911,0.001439991,0.002132165,0.933188],"study_design_scores_gemma":[0.0002279043,0.002100179,0.1077339,0.0002408642,0.0005003179,0.003680235,0.001728801,0.5978217,0.2614484,0.008967546,0.01538187,0.0001684006],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8432229,0.00162878,0.1441882,0.0004065052,0.0001715613,0.000491654,0.002043374,0.002792283,0.005054714],"genre_scores_gemma":[0.7664068,0.0005936916,0.2263287,0.00006350687,0.000154019,0.0001990926,0.003768997,0.00007352187,0.002411578],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006363043,"threshold_uncertainty_score":0.007134616,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2979091142","doi":"10.5539/mas.v13n11p31","title":"Application of Naïve Bayes, Decision Tree, and K-Nearest Neighbors for Automated Text Classification","year":2019,"lang":"en","type":"article","venue":"Modern Applied Science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Computer science; Naive Bayes classifier; Decision tree; k-nearest neighbors algorithm; Artificial intelligence; Precision and recall; Data mining; Tree (set theory); Process (computing); Arabic; Bayes' theorem; Machine learning; The Internet; Translation (biology); Pattern recognition (psychology); Bayesian probability; Mathematics; Support vector machine; World Wide Web","authors":[{"name":"Jafar Ababneh","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01284137028378582,"gpt":0.2646484026858366,"spread":0.2518070324020507,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005361391,0.001369849,0.001846023,0.008593534,0.001432429,0.001938583,0.001374822,0.001217817,0.001649656],"category_scores_gemma":[0.01332682,0.0003637584,0.001205612,0.004395498,0.0004975968,0.002652583,0.0006058095,0.0008904252,0.00106421],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001459729,"about_ca_system_score_gemma":0.002398561,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01770051,"about_ca_topic_score_gemma":0.01516675,"domain_scores_codex":[0.9921831,0.002294926,0.001118688,0.00125177,0.002816783,0.0003346499],"domain_scores_gemma":[0.9918207,0.004195985,0.0005923699,0.0004146505,0.002750009,0.0002263513],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0005047006,0.0005561052,0.01768425,0.000611816,0.0004081558,0.0002248175,0.0002871626,0.03871,0.002486401,0.004024577,0.01001516,0.9244869],"study_design_scores_gemma":[0.00008943775,0.0002439003,0.006524532,0.0001842748,0.0002107235,0.0003644927,0.0004403116,0.9667809,0.005709525,0.0113656,0.007995998,0.0000903145],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2074792,0.01300249,0.7557846,0.001824535,0.001481447,0.001172529,0.002904157,0.004196327,0.01215464],"genre_scores_gemma":[0.5089154,0.00202359,0.4818909,0.0002743541,0.0003671884,0.0003993684,0.002538235,0.00009483755,0.003496186],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01770051,"threshold_uncertainty_score":0.03519499,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3009458272","doi":"10.7717/peerj-cs.261","title":"An evolutionary decomposition-based multi-objective feature selection for multi-label classification","year":2020,"lang":"en","type":"article","venue":"PeerJ Computer Science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Ontario Tech University","funders":"","keywords":"Computer science; Feature selection; Multi-label classification; Artificial intelligence; Feature (linguistics); Data mining; Selection (genetic algorithm); Field (mathematics); Machine learning; Evolutionary algorithm; Genetic algorithm; Pattern recognition (psychology); Optimization problem; Mathematics; Algorithm","authors":[{"name":"Azam Asilian Bidgoli","is_ca":false},{"name":"Hossein Ebrahimpour-Komleh","is_ca":false},{"name":"Shahryar Rahnamayan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0639818680707189,"gpt":0.3394460651024501,"spread":0.2754641970317312,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001274863,0.001088845,0.001155233,0.001258896,0.0005462952,0.0007312284,0.001163062,0.001161638,0.001268466],"category_scores_gemma":[0.002019915,0.0003989477,0.001104389,0.001040672,0.0004309464,0.0008185589,0.000754634,0.0009388037,0.0002037675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000655686,"about_ca_system_score_gemma":0.001005855,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003291526,"about_ca_topic_score_gemma":0.002516288,"domain_scores_codex":[0.999419,0.0001690962,0.00003032971,0.0001152687,0.0001972071,0.00006907945],"domain_scores_gemma":[0.9995541,0.0001773412,0.00004836792,0.00002980884,0.0001615624,0.00002878493],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009372093,0.0001762336,0.001907155,0.0001031212,0.0001152488,0.0001606145,0.0001112695,0.738354,0.01096724,0.004978894,0.002362756,0.2406697],"study_design_scores_gemma":[0.00001003417,0.00004063279,0.0002157081,0.000005779164,0.000009373391,0.00002271693,0.000006985217,0.9980499,0.0007590796,0.0005223913,0.0003523447,0.000005168703],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02623051,0.0002653791,0.971686,0.0001555592,0.00004490123,0.00007689248,0.00003374545,0.0002370878,0.00126989],"genre_scores_gemma":[0.3995241,0.0002501175,0.5964759,0.0002207215,0.00004868535,0.0004521433,0.0002852858,0.00008645084,0.002656749],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003291526,"threshold_uncertainty_score":0.006742239,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W49807964","doi":"","title":"A simple feature selection method for text classification","year":2001,"lang":"en","type":"article","venue":"International Joint Conference on Artificial Intelligence","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université Laval","funders":"","keywords":"Feature selection; Computer science; Information gain; Feature (linguistics); Simple (philosophy); Set (abstract data type); Artificial intelligence; Selection (genetic algorithm); Pattern recognition (psychology); Data mining; Feature extraction; Machine learning","authors":[{"name":"Pascal Soucy","is_ca":true},{"name":"Guy W. Mineau","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1510900227271396,"gpt":0.3835078255874474,"spread":0.2324178028603078,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002278403,0.001651373,0.002019338,0.003022626,0.0009101505,0.00107601,0.001194247,0.001332577,0.004820641],"category_scores_gemma":[0.006156411,0.0003193792,0.001582974,0.003925041,0.0005095998,0.001469815,0.0007955912,0.0009734111,0.003160294],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004149337,"about_ca_system_score_gemma":0.0007573023,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001470541,"about_ca_topic_score_gemma":0.001394637,"domain_scores_codex":[0.9971007,0.000722519,0.0002365316,0.0005252296,0.001292128,0.0001229094],"domain_scores_gemma":[0.9979283,0.001051132,0.0001157213,0.0002287362,0.0006250978,0.00005101961],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002489852,0.0001660993,0.001129406,0.0002850081,0.000159847,0.0001447241,0.00007858877,0.009971893,0.0203154,0.002003688,0.0086727,0.9568236],"study_design_scores_gemma":[0.0005533823,0.001512734,0.01445585,0.0001822656,0.0004831048,0.002662949,0.0001805051,0.7893302,0.08213478,0.02697795,0.08112256,0.0004037596],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.008176571,0.0006047228,0.9860548,0.0001626072,0.0002029307,0.000689494,0.0006146905,0.002649612,0.0008446968],"genre_scores_gemma":[0.07247215,0.0004311839,0.9202034,0.0001619129,0.0002463771,0.00141054,0.001618347,0.0001887184,0.00326735],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004820641,"threshold_uncertainty_score":0.01612669,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2808417254","doi":"10.24963/ijcai.2018/395","title":"Does Tail Label Help for Large-Scale Multi-Label Learning","year":2018,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Novelis (Canada)","funders":"Fundamental Research Funds for the Central Universities; National Natural Science Foundation of China","keywords":"Trimming; Computer science; Multi-label classification; Artificial intelligence; Scale (ratio); Machine learning; Learning to rank; Fraction (chemistry); Pattern recognition (psychology); Ranking (information retrieval)","authors":[{"name":"Tong Wei","is_ca":true},{"name":"Yu-Feng Li","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03686393115797922,"gpt":0.3014385748661733,"spread":0.2645746437081941,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005124908,0.001564801,0.001777216,0.0008514889,0.001453638,0.002632777,0.003533649,0.003368834,0.005243871],"category_scores_gemma":[0.02406773,0.000694151,0.001157362,0.001358887,0.001732366,0.008874291,0.003147439,0.004014139,0.004972439],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001463966,"about_ca_system_score_gemma":0.002420388,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00482297,"about_ca_topic_score_gemma":0.007432547,"domain_scores_codex":[0.9973778,0.0008822563,0.0001104306,0.0008374783,0.0005242245,0.000267818],"domain_scores_gemma":[0.9853855,0.006301773,0.0007373775,0.005333535,0.001562023,0.0006797861],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001731197,0.00118005,0.009881847,0.0004013354,0.0001854437,0.0002217588,0.0005384138,0.2116773,0.01012794,0.02675603,0.0428654,0.6944333],"study_design_scores_gemma":[0.00007838062,0.00009896405,0.0004475855,0.00003396541,0.00002962004,0.00005634356,0.0001298098,0.9455844,0.003631311,0.04648237,0.003401369,0.00002588601],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06298198,0.001058882,0.9164935,0.0044187,0.000437399,0.0002048177,0.0004834638,0.01046095,0.003460334],"genre_scores_gemma":[0.6224774,0.000560612,0.3668726,0.002116123,0.0004811129,0.0002591845,0.001887463,0.0009683546,0.004377101],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005243871,"threshold_uncertainty_score":0.02710342,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4327499178","doi":"10.1007/978-3-031-28241-6","title":"Advances in Information Retrieval","year":2023,"lang":"en","type":"book","venue":"Lecture notes in computer science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Leibniz-Gemeinschaft; University of Massachusetts Amherst; City, University of London; National Institute of Standards and Technology; La Rochelle Université; National Institute of Informatics; Indian Institute of Science Education and Research Mohali; University of Chinese Academy of Sciences; National Institute of Technology Hamirpur; Université de Lyon; Masarykova Univerzita; Ural Federal University; Université de Nantes; Stockholms Universitet; Institut National des Sciences Appliquées de Lyon; Universität Duisburg-Essen; Università di Pisa; Technische Universität Berlin; Bauhaus-Universität Weimar; Universidad de Granada; Università di Bologna; Universität Regensburg; Universitetet i Bergen; University of Windsor; Technische Universität Wien; Radboud Universiteit; University of Waterloo; Commonwealth Scientific and Industrial Research Organisation; Orta Doğu Teknik Üniversitesi; Instituto Superior Técnico; Université de Bretagne Occidentale; Tsinghua University; Chinese Academy of Sciences; Birkbeck, University of London; Universidade de Lisboa; Renmin University of China; Universität Innsbruck; Technische Universiteit Delft; Région Normandie; University of Minnesota; TU Graz, Internationale Beziehungen und Mobilitätsprogramme; Universiteit Utrecht; RMIT University; Haute école Spécialisée de Suisse Occidentale; University of Southampton; Universidade da Coruña; Istituto di Scienza e Tecnologie dell'Informazione; Universidad Nacional de Educación a Distancia; Universiteit Leiden; York University; University of Glasgow; Universität Passau; Universidade do Porto; Dublin City University; Politecnico di Torino; Friedrich-Schiller-Universität Jena; Universitetet i Stavanger; University of Regina; Université Grenoble Alpes; University of Southern Maine; Centre National de la Recherche Scientifique; University of Roehampton; Norges Teknisk-Naturvitenskapelige Universitet; Sun Yat-sen University; Universiteit van Amsterdam; Universidade de Santiago de Compostela; Università degli Studi di Cagliari; Aix-Marseille Université; University of Wolverhampton; Universidade Federal de Minas Gerais; Chinese University of Hong Kong; Aalborg Universitet; Kaiser Permanente; Indian Institute of Technology Kharagpur; Indian Institute of Science; Qatar University; Universidad Autónoma de Madrid; University of Tsukuba; Georgetown University; Universitatea din București; Università degli Studi di Padova; Zürcher Hochschule für Angewandte Wissenschaften; Johns Hopkins University; Indian National Science Academy","keywords":"Computer science; Information retrieval; State (computer science); Artificial intelligence; Programming language","authors":[{"name":"Jaap Kamps","is_ca":false},{"name":"Lorraine Goeuriot","is_ca":false},{"name":"Fábio Crestani","is_ca":false},{"name":"Maria Maistro","is_ca":false},{"name":"Hideo Joho","is_ca":false},{"name":"Brian Davis","is_ca":false},{"name":"Cathal Gurrin","is_ca":false},{"name":"Udo Kruschwitz","is_ca":false},{"name":"Annalina Caputo","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01286416010257471,"gpt":0.2606042818289115,"spread":0.2477401217263368,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001686567,0.001583751,0.002260586,0.005485951,0.0008260285,0.004612012,0.001550185,0.00145425,0.05553482],"category_scores_gemma":[0.003275065,0.0006734013,0.000708312,0.008301434,0.001226128,0.007467553,0.001869412,0.002535229,0.07511046],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001173764,"about_ca_system_score_gemma":0.001288435,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009427242,"about_ca_topic_score_gemma":0.00157018,"domain_scores_codex":[0.9986004,0.000180513,0.00009716572,0.0001955291,0.0008400989,0.00008629448],"domain_scores_gemma":[0.9984585,0.0005228143,0.00007426822,0.0003245224,0.0005075905,0.0001122099],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002672998,0.00004912998,0.00009760056,0.0009403209,0.00003138429,0.00003322029,0.00007417409,0.0003066723,0.001841202,0.04002342,0.3189469,0.6376292],"study_design_scores_gemma":[0.000007540125,0.00004100064,0.0003149337,0.0002754563,0.0000282599,0.0002562537,0.00004054021,0.001160666,0.001029544,0.02804129,0.9687889,0.00001566541],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"other","genre_scores_codex":[0.002054899,0.6013346,0.07706387,0.00759361,0.01659062,0.0001556802,0.0009657664,0.003014821,0.2912262],"genre_scores_gemma":[0.01549842,0.2844719,0.06282778,0.003391319,0.01176666,0.0001458356,0.002168096,0.0009496012,0.6187804],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.05553482,"threshold_uncertainty_score":0.1857824,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3108841339","doi":"10.1109/tcyb.2020.3032707","title":"Semisupervised Learning via Axiomatic Fuzzy Set Theory and SVM","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Cybernetics","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Interpretability; Artificial intelligence; Computer science; Support vector machine; Machine learning; Fuzzy logic; Exploit; Classifier (UML); Schema (genetic algorithms); Set (abstract data type); Frame (networking); Natural language processing; Data mining","authors":[{"name":"Wenjuan Jia","is_ca":false},{"name":"Xiaodong Liu","is_ca":false},{"name":"Yuangang Wang","is_ca":false},{"name":"Witold Pedrycz","is_ca":true},{"name":"Juxiang Zhou","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02293823256316044,"gpt":0.2413329743429583,"spread":0.2183947417797979,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003257115,0.0006852608,0.0009202171,0.001505958,0.0005634181,0.001035963,0.001523149,0.0009944248,0.001031977],"category_scores_gemma":[0.00873274,0.0003057273,0.0007710488,0.001120614,0.001342999,0.002362261,0.001236509,0.001383177,0.0003155982],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008443556,"about_ca_system_score_gemma":0.0009804743,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009470487,"about_ca_topic_score_gemma":0.001130034,"domain_scores_codex":[0.996946,0.001612304,0.000189463,0.0004301519,0.0007467609,0.00007531976],"domain_scores_gemma":[0.994652,0.003169955,0.0005667828,0.0004600774,0.001065777,0.00008540127],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001716878,0.000193646,0.002664995,0.0005337705,0.0002120829,0.0002489141,0.0005242964,0.39232,0.006983561,0.2456683,0.005324566,0.3451542],"study_design_scores_gemma":[0.000007251058,0.00003210093,0.0001369255,0.00001415566,0.000009553942,0.00006104276,0.00002320174,0.942792,0.001151543,0.05488894,0.0008719357,0.00001138002],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004282734,0.0001279062,0.9946941,0.0001253214,0.00001533049,0.00002522169,0.0000249616,0.00009377056,0.0006106045],"genre_scores_gemma":[0.3739579,0.0004071316,0.6230053,0.0002605314,0.0001717009,0.0002609477,0.0003729489,0.00006221217,0.001501394],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003257115,"threshold_uncertainty_score":0.01722544,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}