{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":514,"total_is_capped":false,"direct_labels_cover":5,"predictions_cover":514,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"0eee3839674e","filters":{"topic":"Computational and Text Analysis Methods"}},"results":[{"id":"W2943121181","doi":"10.5465/annals.2017.0099","title":"Topic Modeling in Management Research: Rendering New Theory from Textual Data","year":2019,"lang":"en","type":"article","venue":"Academy of Management Annals","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":591,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto; University of Alberta","funders":"","keywords":"Rendering (computer graphics); Computer science; Novelty; Data science; Grounded theory; Topic model; Artificial intelligence; Sociology; Qualitative research; Social science","authors":[{"name":"Timothy R. Hannigan","is_ca":true},{"name":"Richard Franciscus Johannes Haans","is_ca":false},{"name":"Keyvan Vakili","is_ca":false},{"name":"Hovig Tchalian","is_ca":false},{"name":"Vern Glaser","is_ca":true},{"name":"Milo Shaoqing Wang","is_ca":true},{"name":"Sarah Kaplan","is_ca":true},{"name":"P. Devereaux Jennings","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.5970884422744959,"gpt":0.5250309813464068,"spread":0.07205746092808907,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05415823,0.001470435,0.001731957,0.0186592,0.003416422,0.02001383,0.003454909,0.003480484,0.005457559],"category_scores_gemma":[0.1263661,0.001438459,0.002504147,0.01437381,0.01467233,0.03321538,0.007976783,0.00619147,0.001257444],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005078131,"about_ca_system_score_gemma":0.004467459,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002061707,"about_ca_topic_score_gemma":0.002238594,"domain_scores_codex":[0.9589512,0.03304415,0.001917379,0.002270031,0.003424558,0.0003925643],"domain_scores_gemma":[0.7595305,0.2196856,0.004029687,0.01276293,0.003013552,0.0009776938],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0000803599,0.00008696778,0.003068463,0.001537958,0.0001245999,0.0001692731,0.02971963,0.005771447,0.001055462,0.7911564,0.006711671,0.1605178],"study_design_scores_gemma":[0.00003557571,0.00003084517,0.0007969753,0.0008769651,0.00004758886,0.000123012,0.005452452,0.02704313,0.0007103164,0.9215944,0.04322448,0.00006423538],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00616895,0.004242915,0.9692159,0.01198729,0.0004577546,0.000270204,0.000402516,0.0005543418,0.006700053],"genre_scores_gemma":[0.1725689,0.007409117,0.8102412,0.002072486,0.00220233,0.001842653,0.0007263532,0.0005852224,0.002351915],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9458418,"threshold_uncertainty_score":0.2864195,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3171347074","doi":"10.21105/joss.03272","title":"academictwitteR: an R package to access the Twitter Academic Research Product Track v2 API endpoint","year":2021,"lang":"en","type":"article","venue":"The Journal of Open Source Software","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":227,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Track (disk drive); Product (mathematics); World Wide Web; Computer science; Operating system; Mathematics","authors":[{"name":"Christopher Barrie","is_ca":false},{"name":"Justin Chun‐ting Ho","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.282066190315972,"gpt":0.5272027382753877,"spread":0.2451365479594156,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.00628144,0.003275845,0.002855995,0.00429068,0.001233193,0.004350898,0.003684193,0.001659608,0.1866102],"category_scores_gemma":[0.05428039,0.002233077,0.003416027,0.004617186,0.0008789285,0.003740787,0.005107522,0.003483381,0.2120972],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009105387,"about_ca_system_score_gemma":0.003636778,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005821177,"about_ca_topic_score_gemma":0.009887083,"domain_scores_codex":[0.9965682,0.001213859,0.0003883344,0.0007208179,0.0008386268,0.0002700914],"domain_scores_gemma":[0.9853284,0.008834274,0.0012435,0.001900949,0.002080008,0.0006128912],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000357049,0.00002783964,0.00281061,0.001792408,0.0005642961,0.0001519391,0.0002166949,0.001049122,0.001013888,0.002923045,0.9626884,0.02640473],"study_design_scores_gemma":[0.0006476361,0.0001301355,0.007573267,0.0007304232,0.0007955644,0.0003942373,0.000189105,0.01051234,0.004187279,0.03943875,0.93506,0.0003413568],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"software","genre_scores_codex":[0.002327363,0.001322297,0.1121853,0.001641445,0.001080859,0.0006518287,0.5341508,0.3334641,0.01317598],"genre_scores_gemma":[0.02508354,0.00152413,0.1789288,0.00282239,0.0005833006,0.006954219,0.3886084,0.3679588,0.02753647],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.9963158,"threshold_uncertainty_score":0.6242734,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2406594196","doi":"10.5121/csit.2016.60616","title":"A Text Mining Research Based on LDA Topic Modelling","year":2016,"lang":"en","type":"article","venue":"","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":174,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Acadia University","funders":"","keywords":"Computer science; Topic model; Natural language processing; Artificial intelligence; Information retrieval","authors":[{"name":"Tong Zhou","is_ca":true},{"name":"Haiyi Zhang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3539824256636802,"gpt":0.5065979207347818,"spread":0.1526154950711016,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003735346,0.001265021,0.001440599,0.007565933,0.001369848,0.003936142,0.001535406,0.001381179,0.001871652],"category_scores_gemma":[0.01092167,0.0006263517,0.002077804,0.01015437,0.001001733,0.006663283,0.001218205,0.002012996,0.002445176],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001099215,"about_ca_system_score_gemma":0.00132101,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002475651,"about_ca_topic_score_gemma":0.001864415,"domain_scores_codex":[0.9968035,0.001175576,0.0002147347,0.0009044542,0.0007817782,0.0001199813],"domain_scores_gemma":[0.9944542,0.003498274,0.0003148789,0.0005867136,0.001032457,0.0001135189],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002614462,0.0004134879,0.01090494,0.001860013,0.0004859987,0.0003331419,0.001796459,0.02122093,0.009991318,0.06975954,0.02095594,0.8620168],"study_design_scores_gemma":[0.00007406251,0.0003347021,0.009266604,0.0005778982,0.0004038217,0.001360536,0.001575888,0.724333,0.01511658,0.1362688,0.1104195,0.0002686492],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01472545,0.008525891,0.9659724,0.002081823,0.0005587631,0.00025028,0.0009532176,0.001378529,0.005553575],"genre_scores_gemma":[0.3325682,0.01866843,0.6264296,0.001004752,0.00267263,0.001128232,0.003934827,0.0005970519,0.01299623],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007565933,"threshold_uncertainty_score":0.01975465,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2041314074","doi":"10.1017/s0007123411000160","title":"Language and Ideology in Congress","year":2011,"lang":"en","type":"article","venue":"British Journal of Political Science","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":168,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal; Kellogg's (Canada)","funders":"","keywords":"Ideology; Legislature; Interpretation (philosophy); Political science; Politics; Law; Linguistics; Philosophy","authors":[{"name":"Daniel Diermeier","is_ca":true},{"name":"Jean-François Godbout","is_ca":true},{"name":"Bei Yu","is_ca":false},{"name":"Stefan Kaufmann","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03988064428790917,"gpt":0.3808767252756357,"spread":0.3409960809877265,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001724147,0.000137602,0.0002723724,0.004197621,0.001067097,0.003095388,0.0002103056,0.0003658635,0.006268361],"category_scores_gemma":[0.01356138,0.0001077635,0.0001818311,0.005410894,0.001282297,0.001013277,0.001291846,0.0006194997,0.0009624853],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008814016,"about_ca_system_score_gemma":0.0006290037,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002458527,"about_ca_topic_score_gemma":0.004007597,"domain_scores_codex":[0.9975016,0.001248173,0.0001962179,0.000246939,0.0005660599,0.0002411449],"domain_scores_gemma":[0.9893234,0.006638985,0.002695382,0.0002389617,0.0007750483,0.000328093],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0009254325,0.0003774277,0.6649885,0.0006516337,0.0002486448,0.002371716,0.1004723,0.001316938,0.004879212,0.02614229,0.01215539,0.1854705],"study_design_scores_gemma":[0.00001870508,0.0001049474,0.9375725,0.0003307005,0.00004468232,0.0006359175,0.03022526,0.001046854,0.0005850993,0.003411992,0.02597471,0.00004870081],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.978255,0.001674104,0.0002300751,0.0008862386,0.0000572876,0.000008511232,0.0005658733,0.000009103448,0.01831394],"genre_scores_gemma":[0.9971994,0.0005102429,0.0001847361,0.00006578738,0.00005834156,0.00001593692,0.0004007908,0.0000117339,0.001553026],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006268361,"threshold_uncertainty_score":0.02096975,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2954211942","doi":"10.1017/pan.2019.26","title":"Word Embeddings for the Analysis of Ideological Placement in Parliamentary Corpora","year":2019,"lang":"en","type":"article","venue":"Political Analysis","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":165,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Toronto","funders":"","keywords":"Ideology; Metadata; Parliament; Computer science; Context (archaeology); Politics; Word (group theory); Natural language processing; Language model; Linguistics; Artificial intelligence; Data science; Sociology; Political science; World Wide Web; Law; History","authors":[{"name":"Ludovic Rheault","is_ca":true},{"name":"Christopher Cochrane","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0491700587411518,"gpt":0.3975014255878185,"spread":0.3483313668466667,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002981274,0.0007623621,0.0004066837,0.002955179,0.00082577,0.001870104,0.0007007482,0.0008099245,0.003794079],"category_scores_gemma":[0.02239585,0.0003959334,0.0007372224,0.004168221,0.0009957369,0.004091859,0.001663432,0.002218788,0.001781096],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008928708,"about_ca_system_score_gemma":0.0007929655,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002536083,"about_ca_topic_score_gemma":0.005868912,"domain_scores_codex":[0.9980797,0.001132824,0.0001512293,0.0003352782,0.0002284696,0.00007259264],"domain_scores_gemma":[0.9911631,0.006090297,0.0008938746,0.001133243,0.0005753563,0.0001441773],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004957747,0.0005202418,0.04287893,0.0007385371,0.0004439398,0.0002946585,0.003060259,0.1531495,0.01085142,0.1101356,0.0191697,0.6582614],"study_design_scores_gemma":[0.00003442177,0.0001239651,0.0111189,0.0001198195,0.00004789968,0.0001786697,0.0007654727,0.8445358,0.002889937,0.1270358,0.01309176,0.00005752261],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1465732,0.000847662,0.8410559,0.0008302015,0.0001927382,0.0001797388,0.004431641,0.001776456,0.004112416],"genre_scores_gemma":[0.6527936,0.0005202238,0.3331651,0.0001234469,0.0001244327,0.0007691974,0.009010492,0.0004496922,0.003043744],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003794079,"threshold_uncertainty_score":0.01576668,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3000486466","doi":"10.31222/osf.io/5rksu","title":"Open Science Practices are on the Rise: The State of Social Science (3S) Survey","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":137,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Transparency (behavior); Open science; Quarter (Canadian coin); Social science; Sociology; State (computer science); Science studies; Field (mathematics); Political science; Public relations; Geography; Law","authors":[{"name":"Garret Christensen","is_ca":false},{"name":"Zenan Wang","is_ca":false},{"name":"Elizabeth Levy Paluck","is_ca":false},{"name":"Nicholas Swanson","is_ca":false},{"name":"David J. Birke","is_ca":false},{"name":"Edward Miguel","is_ca":false},{"name":"Rebecca Littman","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2535066719527609,"gpt":0.5222285857905022,"spread":0.2687219138377414,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","open_science"],"consensus_categories":[],"category_scores_codex":[0.0129551,0.0001292146,0.0003734115,0.006229216,0.0008720942,0.002999023,0.0006438415,0.0009908358,0.002188453],"category_scores_gemma":[0.04180241,0.0002972449,0.0003801985,0.008399134,0.001534026,0.003226308,0.002763783,0.001294849,0.0005398502],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001892168,"about_ca_system_score_gemma":0.002634557,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008967833,"about_ca_topic_score_gemma":0.009112854,"domain_scores_codex":[0.9895239,0.003672635,0.001482734,0.0009210151,0.003442808,0.0009568899],"domain_scores_gemma":[0.880537,0.04174925,0.04701513,0.005619491,0.01666632,0.008412682],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00006692876,0.00007117501,0.9530813,0.0001816121,0.00005730341,0.00006097393,0.01184844,0.00006629232,0.0003612801,0.001622525,0.004473294,0.02810889],"study_design_scores_gemma":[0.000004602171,0.00006418942,0.9665676,0.0001530983,0.00001315628,0.0001017695,0.01605066,0.0001996963,0.0001709274,0.0005565597,0.01609402,0.00002371695],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.985687,0.0009145759,0.0005216481,0.004161977,0.00003850372,0.00006213329,0.003632149,0.00002485186,0.004957148],"genre_scores_gemma":[0.9949726,0.0008078302,0.0004354031,0.0008622532,0.0000643219,0.00009919141,0.002172005,0.00001533273,0.0005710737],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9993562,"threshold_uncertainty_score":0.06851393,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2900532930","doi":"10.1111/joes.12370","title":"ECONOMETRICS MEETS SENTIMENT: AN OVERVIEW OF METHODOLOGY AND APPLICATIONS","year":2020,"lang":"en","type":"article","venue":"Journal of Economic Surveys","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":137,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Innoviris; Vrije Universiteit Brussel; Institut de Valorisation des Données; European Commission; Universiteit Gent; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Econometrics; Econometric model; Field (mathematics); Econometric analysis; Economics; Sentiment analysis; Computer science; Artificial intelligence; Mathematics","authors":[{"name":"Andres Algaba","is_ca":false},{"name":"David Ardia","is_ca":false},{"name":"Keven Bluteau","is_ca":false},{"name":"Samuel Borms","is_ca":false},{"name":"Kris Boudt","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4794215240349278,"gpt":0.4705475097426656,"spread":0.008874014292262145,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02449385,0.001461653,0.002298718,0.01075716,0.0009430571,0.006344237,0.001816173,0.002861532,0.01151957],"category_scores_gemma":[0.06905397,0.001217894,0.001917537,0.01311682,0.002807296,0.005278169,0.003051683,0.003874474,0.006419611],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001911734,"about_ca_system_score_gemma":0.002941853,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002312619,"about_ca_topic_score_gemma":0.00115314,"domain_scores_codex":[0.9831292,0.0121907,0.001069878,0.0008632361,0.00250372,0.0002432152],"domain_scores_gemma":[0.9389213,0.05139619,0.001853108,0.002912617,0.004302777,0.0006140653],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00005982442,0.0001098107,0.005027909,0.003310491,0.0002796993,0.0002471716,0.0006162326,0.007508876,0.0007260843,0.4404673,0.03800601,0.5036407],"study_design_scores_gemma":[0.00004070922,0.00004584116,0.004223492,0.002051275,0.00005700472,0.0002937764,0.0003553991,0.0340487,0.0005262719,0.7617128,0.1965514,0.00009341042],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.001687563,0.05204581,0.9073076,0.01622155,0.001038334,0.0004287159,0.001291734,0.001038077,0.01894077],"genre_scores_gemma":[0.08571722,0.1172508,0.7681383,0.005378122,0.009365521,0.002689626,0.001849279,0.001181901,0.008429322],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.02449385,"threshold_uncertainty_score":0.1295375,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4391220979","doi":"10.1287/mksc.2023.0454","title":"Frontiers: Determining the Validity of Large Language Models for Automated Perceptual Analysis","year":2024,"lang":"en","type":"article","venue":"Marketing Science","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":131,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Perception; Computer science; Data science; Human language; Natural language processing; Language model; Artificial intelligence; Cognitive psychology; Machine learning; Econometrics; Economics; Psychology; Linguistics","authors":[{"name":"Peiyao Li","is_ca":false},{"name":"Noah Castelo","is_ca":true},{"name":"Zsolt Katona","is_ca":false},{"name":"Miklós Sárváry","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04878391879767375,"gpt":0.4075470857221072,"spread":0.3587631669244334,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01432575,0.001313321,0.0009017388,0.002163933,0.001310867,0.004983549,0.001766333,0.001598563,0.006755377],"category_scores_gemma":[0.1096614,0.0006853343,0.001284355,0.001011143,0.002497862,0.009420702,0.003825568,0.002783271,0.002140907],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001343235,"about_ca_system_score_gemma":0.001764656,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004965283,"about_ca_topic_score_gemma":0.00350942,"domain_scores_codex":[0.9898239,0.007222359,0.0003348776,0.001171556,0.001129301,0.0003180266],"domain_scores_gemma":[0.8459277,0.1406275,0.0021106,0.007041455,0.003202262,0.001090534],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.005386849,0.001587588,0.03422331,0.001412561,0.0006016809,0.001102908,0.007221573,0.1685957,0.03119779,0.1194721,0.0244286,0.6047692],"study_design_scores_gemma":[0.0001405763,0.0002503211,0.002698902,0.00006518636,0.00004771851,0.0001576862,0.000760927,0.9137572,0.003607471,0.07575272,0.002711252,0.00005011332],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2810509,0.0009882619,0.6962207,0.001780699,0.000319998,0.0007059227,0.00192498,0.005604601,0.01140396],"genre_scores_gemma":[0.8772382,0.0001747562,0.119044,0.0002335979,0.0000911457,0.0004129232,0.001390314,0.0004601898,0.0009548303],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01432575,"threshold_uncertainty_score":0.07576275,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2593732251","doi":"10.1017/s0922156517000085","title":"Can Quantitative Methods Complement Doctrinal Legal Studies? Using Citation Network and Corpus Linguistic Analysis to Understand International Courts","year":2017,"lang":"en","type":"article","venue":"Leiden Journal of International Law","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":130,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Centre for International Governance Innovation","funders":"","keywords":"Legal translation; Political science; Empirical legal studies; Law; Jurisprudence; Sources of law; International law; Comparative law; Human rights; Legal culture; Sociology","authors":[{"name":"Urška Šadl","is_ca":true},{"name":"Henrik Palmer Olsen","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2317016934668139,"gpt":0.5415538491311146,"spread":0.3098521556643007,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.03448306,0.000812432,0.001177296,0.02973363,0.001634301,0.01208422,0.001752471,0.00194572,0.005825541],"category_scores_gemma":[0.1503362,0.0003620571,0.0005908354,0.02892317,0.006288969,0.018438,0.003250435,0.001969954,0.000623862],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002688035,"about_ca_system_score_gemma":0.002232525,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002269058,"about_ca_topic_score_gemma":0.002899025,"domain_scores_codex":[0.9662234,0.02637732,0.001896859,0.001569166,0.003652846,0.0002803958],"domain_scores_gemma":[0.7638956,0.2076607,0.009454955,0.009277418,0.008983223,0.0007280391],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.00007382964,0.0001470131,0.01511384,0.002873731,0.0005153807,0.0001667909,0.01176495,0.004050047,0.000732575,0.7229953,0.01379964,0.2277668],"study_design_scores_gemma":[0.00002805113,0.00004790361,0.01269356,0.00212292,0.0001201605,0.0001335435,0.009037684,0.03709529,0.0007968015,0.8651175,0.072707,0.00009959225],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1034651,0.04543164,0.7162409,0.05136638,0.003573218,0.000601507,0.003429803,0.0009979679,0.07489347],"genre_scores_gemma":[0.7539317,0.01272141,0.2231617,0.00193663,0.002417805,0.001111877,0.001323173,0.0002855382,0.003110116],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9702663,"threshold_uncertainty_score":0.182366,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3194827031","doi":"10.1177/14614448211038761","title":"Using social media images as data in social science research","year":2021,"lang":"en","type":"article","venue":"New Media & Society","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":102,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"Social Sciences and Humanities Research Council of Canada; Nova Scotia Research Innovation Trust","keywords":"Social media; Data science; Thematic analysis; Computer science; Social research; Coding (social sciences); Narrative; Thematic map; Sociology; Social science; Qualitative research; World Wide Web","authors":[{"name":"Yan Chen","is_ca":true},{"name":"Kate Sherren","is_ca":true},{"name":"Michael Smit","is_ca":true},{"name":"Kyung Young Lee","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4689358522654855,"gpt":0.5683473242748014,"spread":0.09941147200931594,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06858966,0.00111716,0.002298928,0.0455186,0.00199905,0.009887229,0.002434113,0.003265679,0.004859949],"category_scores_gemma":[0.2692396,0.001249672,0.002696805,0.03266435,0.005502175,0.01454612,0.00577329,0.00294775,0.001086744],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004192139,"about_ca_system_score_gemma":0.01368628,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004365546,"about_ca_topic_score_gemma":0.009526673,"domain_scores_codex":[0.9262983,0.04419745,0.01351308,0.004261367,0.0108788,0.0008510714],"domain_scores_gemma":[0.5046233,0.4421433,0.02026119,0.008754351,0.0230991,0.001118741],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001839604,0.0001022023,0.009181391,0.4382299,0.002012888,0.0004818099,0.01892423,0.00072124,0.001139156,0.04039413,0.01562666,0.4730024],"study_design_scores_gemma":[0.00005738994,0.0001339068,0.009343222,0.651939,0.002947788,0.0004621015,0.01927216,0.0006850447,0.001499474,0.02367759,0.289844,0.0001382501],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"review","genre_gemma":"methods","genre_scores_codex":[0.01673151,0.9131604,0.02232991,0.0199241,0.002803993,0.002582308,0.005008642,0.0001139971,0.01734513],"genre_scores_gemma":[0.1397774,0.7802212,0.05518748,0.008024081,0.001643466,0.009266194,0.003973629,0.0002105302,0.001695995],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9314103,"threshold_uncertainty_score":0.3627411,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1597876521","doi":"10.1002/9781119959847.ch1","title":"“It Looks Great but How do I know if it Fits?”: An Introduction to Meta‐Synthesis Research","year":2011,"lang":"en","type":"other","venue":"","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":76,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Thompson Rivers University","funders":"","keywords":"Selection (genetic algorithm); Computer science; Qualitative research; Management science; Epistemology; Data science; Engineering ethics; Sociology; Engineering; Artificial intelligence; Social science; Philosophy","authors":[{"name":"Barbara Paterson","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2145971558398435,"gpt":0.4432425419013202,"spread":0.2286453860614767,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0531918,0.001424586,0.001851829,0.006083897,0.002036659,0.007146685,0.00246411,0.002968872,0.0196479],"category_scores_gemma":[0.07245659,0.001290392,0.001997586,0.01102579,0.004712136,0.008892049,0.003785722,0.006116908,0.00579706],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004731487,"about_ca_system_score_gemma":0.008537688,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002137816,"about_ca_topic_score_gemma":0.004472563,"domain_scores_codex":[0.9521816,0.04117999,0.002142287,0.001196251,0.003075349,0.0002245803],"domain_scores_gemma":[0.914861,0.0762258,0.001698046,0.002542457,0.004223467,0.0004493148],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0000777366,0.00005766827,0.0002877422,0.01891597,0.0002888103,0.0001286715,0.01363266,0.001437957,0.000655339,0.5162089,0.1686393,0.2796692],"study_design_scores_gemma":[0.00002997412,0.00005222138,0.0004339074,0.01555213,0.00008085014,0.0001766167,0.002749531,0.001044378,0.0006227559,0.1927826,0.7864063,0.00006888617],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001824688,0.2730001,0.5140651,0.08472405,0.01195502,0.006372002,0.003727042,0.001193723,0.1031381],"genre_scores_gemma":[0.01972866,0.1952961,0.7152892,0.01698802,0.00271874,0.01763699,0.001635219,0.0009760683,0.02973104],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9468082,"threshold_uncertainty_score":0.2813085,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3019170642","doi":"10.1017/s0008423920000372","title":"(Un)Covering the COVID-19 Pandemic: Framing Analysis of the Crisis in Canada","year":2020,"lang":"en","type":"article","venue":"Canadian Journal of Political Science","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":73,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Toronto; Université Laval","funders":"","keywords":"Coronavirus disease 2019 (COVID-19); Pandemic; Framing (construction); Political science; 2019-20 coronavirus outbreak; Severe acute respiratory syndrome coronavirus 2 (SARS-CoV-2); Crisis management; Economic history; Geography; History; Outbreak; Law; Medicine; Infectious disease (medical specialty); Virology","authors":[{"name":"William Poirier","is_ca":true},{"name":"Catherine Ouellet","is_ca":true},{"name":"Marc‐Antoine Rancourt","is_ca":true},{"name":"Justine Béchard","is_ca":true},{"name":"Yannick Dufresne","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07415035811355343,"gpt":0.3646452236737246,"spread":0.2904948655601712,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001811165,0.0004020751,0.0003673809,0.004318381,0.005932748,0.004089689,0.0008461403,0.0006445023,0.008623371],"category_scores_gemma":[0.01219495,0.0001973539,0.0004907989,0.01110276,0.001209424,0.001231885,0.001548442,0.00141547,0.001048833],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.05345154,"about_ca_system_score_gemma":0.06027127,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.9964575,"about_ca_topic_score_gemma":0.9972354,"domain_scores_codex":[0.9987275,0.0001924308,0.00004419982,0.0001677616,0.0004923189,0.0003758583],"domain_scores_gemma":[0.9919586,0.001323257,0.0004671814,0.0003570412,0.004837487,0.00105643],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"qualitative","study_design_scores_codex":[0.0004370535,0.0001230367,0.2362872,0.0005922618,0.0002346424,0.0004156263,0.01708306,0.004910451,0.0006555096,0.04734292,0.576393,0.1155252],"study_design_scores_gemma":[0.00006211428,0.00004362411,0.4724302,0.0008249427,0.0002583497,0.0001051986,0.0513902,0.01868933,0.001116136,0.0087986,0.4460674,0.0002138943],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5550822,0.01106687,0.00537983,0.05522622,0.0006424635,0.0004649831,0.2162839,0.0005221463,0.1553314],"genre_scores_gemma":[0.9007872,0.004780126,0.0043888,0.002727789,0.0001599535,0.0001773509,0.05882706,0.000204242,0.02794738],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05345154,"threshold_uncertainty_score":0.3878199,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4251909731","doi":"10.1142/p981","title":"Exploring Big Historical Data","year":2014,"lang":"en","type":"book","venue":"IMPERIAL COLLEGE PRESS eBooks","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":66,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo; Carleton University","funders":"","keywords":"Big data; History; Computer science; Data mining","authors":[{"name":"Shawn Graham","is_ca":true},{"name":"Ian Milligan","is_ca":true},{"name":"Scott Weingart","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.485653745540465,"gpt":0.3824765685860768,"spread":0.1031771769543882,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001028782,0.0007080981,0.0004798847,0.002968564,0.0007305664,0.003987535,0.000750683,0.0006198508,0.02582148],"category_scores_gemma":[0.005673574,0.0006900807,0.0006285438,0.00500387,0.001317612,0.006158476,0.002399328,0.001787628,0.005772765],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007712112,"about_ca_system_score_gemma":0.0009400123,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003360219,"about_ca_topic_score_gemma":0.008417541,"domain_scores_codex":[0.9996371,0.0001040321,0.00002070874,0.00006667495,0.0001549989,0.00001643968],"domain_scores_gemma":[0.9978598,0.001581181,0.00005269113,0.0003104692,0.0001343866,0.00006146134],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0000257773,0.00001667261,0.001164619,0.0005500792,0.00008214542,0.0001837984,0.001445853,0.004045983,0.0004711228,0.1719163,0.2540857,0.5660118],"study_design_scores_gemma":[0.000003120765,0.000006593083,0.001022242,0.0004168895,0.00002224504,0.0001817059,0.0006563041,0.005172792,0.0004129645,0.3657739,0.6263123,0.00001895794],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00794128,0.06002413,0.5229436,0.02053188,0.002648945,0.0001505503,0.01086577,0.004832002,0.3700619],"genre_scores_gemma":[0.1265851,0.08430412,0.438305,0.003539274,0.002341182,0.0004245692,0.02607173,0.003630888,0.3147981],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02582148,"threshold_uncertainty_score":0.08638144,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2567467453","doi":"10.18653/v1/d16-1148","title":"Analyzing Framing through the Casts of Characters in the News","year":2016,"lang":"en","type":"article","venue":"","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":64,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; University of Washington","keywords":"Framing (construction); Computer science; History; Archaeology","authors":[{"name":"Dallas Card","is_ca":false},{"name":"Justin H. Gross","is_ca":false},{"name":"Amber E. Boydstun","is_ca":false},{"name":"Noah A. Smith","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0615801185775051,"gpt":0.3893943235540798,"spread":0.3278142049765747,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008691968,0.0005318408,0.0003520044,0.003524333,0.0007144871,0.001978504,0.0004436219,0.000641068,0.001985082],"category_scores_gemma":[0.00721535,0.0003485037,0.0004176819,0.002593626,0.0006822362,0.001964284,0.0008889862,0.001200454,0.0008835925],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006349609,"about_ca_system_score_gemma":0.0004268106,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006632254,"about_ca_topic_score_gemma":0.01174776,"domain_scores_codex":[0.9992956,0.0002845148,0.00002810697,0.0001825863,0.0001134354,0.00009572828],"domain_scores_gemma":[0.996155,0.00210242,0.0007412348,0.0004630268,0.000365788,0.0001725603],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001453362,0.0006563302,0.3000588,0.0004565949,0.0003493757,0.00107775,0.009581422,0.0985759,0.03932543,0.07471886,0.01535851,0.4583878],"study_design_scores_gemma":[0.00003550606,0.0001389497,0.146312,0.0001047648,0.0001590497,0.0003756283,0.002900353,0.7637655,0.01331022,0.0538444,0.01896059,0.0000930293],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7446055,0.0008357149,0.2359716,0.0008315014,0.00009868915,0.00007696188,0.002775282,0.0009691893,0.01383553],"genre_scores_gemma":[0.9736623,0.0002194947,0.0216476,0.000042676,0.00008128535,0.00005159514,0.002111511,0.00008796505,0.002095681],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006632254,"threshold_uncertainty_score":0.01318735,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2917452130","doi":"10.1002/wcc.576","title":"Frontiers in data analytics for adaptation research: Topic modeling","year":2019,"lang":"en","type":"article","venue":"Wiley Interdisciplinary Reviews Climate Change","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":64,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true},"ca_institutions":"McGill University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Adaptation (eye); Corporate governance; Leverage (statistics); Data science; Topic model; Convention; Climate change adaptation; Vulnerability (computing); Political science; Climate change; Computer science; Sociology; Social science; Business; Ecology","authors":[{"name":"Alexandra Lesnikowski","is_ca":true},{"name":"Ella Belfer","is_ca":true},{"name":"Emma Rodman","is_ca":false},{"name":"Julie Smith","is_ca":false},{"name":"Robbert Biesbroek","is_ca":false},{"name":"John Wilkerson","is_ca":false},{"name":"James D. Ford","is_ca":false},{"name":"Lea Berrang‐Ford","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.661302599603033,"gpt":0.5346772663937048,"spread":0.1266253332093281,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03890215,0.001682846,0.002961196,0.01158172,0.001667293,0.01373036,0.003628171,0.003242004,0.003924277],"category_scores_gemma":[0.1113164,0.001063831,0.002869233,0.01848955,0.004564946,0.01586477,0.005586138,0.007499276,0.00203466],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003351198,"about_ca_system_score_gemma":0.004529995,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004205671,"about_ca_topic_score_gemma":0.003364885,"domain_scores_codex":[0.9718868,0.02039286,0.001637499,0.003207623,0.002626847,0.0002483329],"domain_scores_gemma":[0.7860616,0.1942783,0.004333099,0.01010074,0.004078287,0.001147945],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001726882,0.000307222,0.02471119,0.004218298,0.0008667624,0.0003117254,0.005268088,0.0288584,0.00127292,0.4433542,0.0378102,0.4528484],"study_design_scores_gemma":[0.00003122153,0.00004580056,0.003545511,0.0009542344,0.00007062948,0.0001512456,0.001950243,0.1688802,0.0004236837,0.7834033,0.04045489,0.00008896903],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.008556634,0.02781768,0.9090545,0.04092802,0.0006705133,0.0005237316,0.004900708,0.001095615,0.006452637],"genre_scores_gemma":[0.2178002,0.02889227,0.7327695,0.004133839,0.004397281,0.002240553,0.007078578,0.0004827612,0.002204964],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03890215,"threshold_uncertainty_score":0.2057367,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3165669050","doi":"10.1007/s11135-021-01164-0","title":"Algorithmic thinking in the public interest: navigating technical, legal, and ethical hurdles to web scraping in the social sciences","year":2021,"lang":"en","type":"article","venue":"Quality & Quantity","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":60,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Winnipeg; Carleton University; University of Toronto","funders":"","keywords":"Context (archaeology); Openness to experience; The Internet; Data sharing; Social media; Computer science; Principle of legality; World Wide Web; Data science; Internet privacy; Knowledge management; Public relations; Engineering ethics; Political science; Engineering; Psychology; Law","authors":[{"name":"Alex Luscombe","is_ca":true},{"name":"Kevin Dick","is_ca":true},{"name":"Kevin Walby","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.279075911413793,"gpt":0.5367740553267564,"spread":0.2576981439129634,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.08383428,0.0005973809,0.001203139,0.006321489,0.005979401,0.01936684,0.003952756,0.005327185,0.01046392],"category_scores_gemma":[0.2861557,0.0009097514,0.001343031,0.004499527,0.03108607,0.02811372,0.008633071,0.009017224,0.001481705],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004481817,"about_ca_system_score_gemma":0.01104072,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00396925,"about_ca_topic_score_gemma":0.006248583,"domain_scores_codex":[0.9285536,0.05830081,0.001857417,0.002653089,0.007790811,0.0008442312],"domain_scores_gemma":[0.5693239,0.3831802,0.01026737,0.02115748,0.01344081,0.002630275],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000083898,0.0001612127,0.005458554,0.0004293912,0.0001015225,0.0001236919,0.01045789,0.004834611,0.0004078268,0.866368,0.009465519,0.1021078],"study_design_scores_gemma":[0.00001248843,0.000007948545,0.0004066951,0.0001787044,0.00001145011,0.0000331728,0.001533542,0.00666421,0.0002213878,0.9859553,0.004959841,0.00001524576],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07016579,0.004313372,0.6990159,0.1747147,0.000794485,0.0003029348,0.0002956435,0.0009451141,0.04945202],"genre_scores_gemma":[0.6546041,0.002025924,0.3264698,0.009824807,0.001045191,0.0005609469,0.0002517806,0.0007873286,0.004430069],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9161657,"threshold_uncertainty_score":0.4433634,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2578588058","doi":"10.1017/s0008423916001165","title":"Digitization of the Canadian Parliamentary Debates","year":2017,"lang":"en","type":"article","venue":"Canadian Journal of Political Science","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":56,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Toronto","funders":"","keywords":"Digitization; House of Commons; Political science; Architecture; Politics; Commons; Public administration; Library science; Media studies; Computer science; Sociology; Geography; Law; Telecommunications; Parliament; Archaeology","authors":[{"name":"Kaspar Beelen","is_ca":false},{"name":"Timothy Alberdingk Thijm","is_ca":true},{"name":"Christopher Cochrane","is_ca":true},{"name":"Kees Halvemaan","is_ca":false},{"name":"Graeme Hirst","is_ca":true},{"name":"Michael Kimmins","is_ca":true},{"name":"Sander Lijbrink","is_ca":false},{"name":"M. Marx","is_ca":false},{"name":"Nona Naderi","is_ca":true},{"name":"Ludovic Rheault","is_ca":true},{"name":"Roman Polyanovsky","is_ca":true},{"name":"Tanya Whyte","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04302063556107077,"gpt":0.3648532980313726,"spread":0.3218326624703018,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005077076,0.0005124256,0.0005949807,0.03112714,0.01040064,0.007858695,0.002159298,0.000698077,0.01927991],"category_scores_gemma":[0.02124201,0.0005188895,0.0004914409,0.05680427,0.00385233,0.002139044,0.00470705,0.001820757,0.001354613],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.1010198,"about_ca_system_score_gemma":0.1204231,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.9830294,"about_ca_topic_score_gemma":0.9907219,"domain_scores_codex":[0.9922996,0.000698328,0.0002957454,0.0009130543,0.004726343,0.001066992],"domain_scores_gemma":[0.9835872,0.002627538,0.001006349,0.002406132,0.009381157,0.0009916053],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002663651,0.00004397941,0.02684791,0.001357719,0.00008202263,0.0005867568,0.05895795,0.002318987,0.002226836,0.1466528,0.2465142,0.5141445],"study_design_scores_gemma":[0.000007339689,0.000005987213,0.0527788,0.0004545281,0.00003688919,0.0001132841,0.005993109,0.0003053505,0.001174382,0.001848808,0.937219,0.00006251618],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.1929832,0.02245297,0.01250826,0.01661532,0.001715032,0.0009415001,0.1929654,0.001278863,0.5585393],"genre_scores_gemma":[0.758527,0.01451429,0.02785986,0.00211036,0.0004903103,0.0008148491,0.07882314,0.0007846752,0.1160755],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1010198,"threshold_uncertainty_score":0.7329537,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2791572729","doi":"10.1073/pnas.1719792115","title":"Measuring discursive influence across scholarship","year":2018,"lang":"en","type":"article","venue":"Proceedings of the National Academy of Sciences","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":54,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Air Force Office of Scientific Research; Killam Trusts; John Templeton Foundation; National Science Foundation","keywords":"Scholarship; TRACE (psycholinguistics); Citation; Sociology; Data science; Measure (data warehouse); Epistemology; Social science; Computer science; Political science; Library science; Linguistics; Data mining","authors":[{"name":"Aaron Gerow","is_ca":false},{"name":"Yuening Hu","is_ca":false},{"name":"Jordan Boyd‐Graber","is_ca":false},{"name":"David M. Blei","is_ca":false},{"name":"James A. Evans","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1460954563481362,"gpt":0.4331857178187166,"spread":0.2870902614705804,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.01458354,0.000494402,0.001065327,0.0216092,0.002134267,0.006020596,0.001244395,0.00117666,0.002154168],"category_scores_gemma":[0.1214645,0.0003845189,0.0007508687,0.02398769,0.002337921,0.006832541,0.004459736,0.001063875,0.0007849063],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002337619,"about_ca_system_score_gemma":0.001724284,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004148101,"about_ca_topic_score_gemma":0.005060721,"domain_scores_codex":[0.9863179,0.003803515,0.001484291,0.002290968,0.005520363,0.0005829476],"domain_scores_gemma":[0.8819938,0.07303789,0.02125379,0.008249162,0.01240042,0.003064896],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002458637,0.0003455436,0.8181382,0.0006544677,0.0007210689,0.000213112,0.02205992,0.005362023,0.004971917,0.01626622,0.00196446,0.1290571],"study_design_scores_gemma":[0.00003456865,0.0003750329,0.9309677,0.0001536789,0.0003096612,0.0002699767,0.006183817,0.01309822,0.005407016,0.02686215,0.01622567,0.0001124685],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9609146,0.001993207,0.0155653,0.0006170002,0.00005595228,0.0002604221,0.001766198,0.0001552753,0.01867212],"genre_scores_gemma":[0.9905749,0.0004773354,0.005802898,0.00006927154,0.0001825471,0.0002875731,0.001098024,0.00005949581,0.001447927],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9783908,"threshold_uncertainty_score":0.07712609,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2257785951","doi":"10.1016/j.shpsc.2015.12.016","title":"What are narratives good for?","year":2016,"lang":"en","type":"article","venue":"Studies in History and Philosophy of Science Part C Studies in History and Philosophy of Biological and Biomedical Sciences","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":52,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Narrative; Contingency; Narrative criticism; Narrative network; Narrative inquiry; Natural (archaeology); Epistemology; History; Sociology; Aesthetics; Literature; Art; Philosophy","authors":[{"name":"John Beatty","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3121875705006407,"gpt":0.4168443495923639,"spread":0.1046567790917232,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01411811,0.0009007938,0.001093335,0.004632466,0.01241333,0.02796738,0.001841682,0.00556838,0.01434735],"category_scores_gemma":[0.05328663,0.0005230306,0.0004271507,0.005009769,0.04833294,0.05108801,0.00561632,0.00632435,0.001868682],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009058119,"about_ca_system_score_gemma":0.007660624,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006228726,"about_ca_topic_score_gemma":0.006674569,"domain_scores_codex":[0.9847575,0.0114301,0.0004557199,0.0009198665,0.001542859,0.0008938451],"domain_scores_gemma":[0.9582054,0.02932285,0.003691886,0.002994257,0.002981126,0.002804485],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002378092,0.000009676824,0.001060615,0.0001742535,0.00001519437,0.0001117725,0.07481726,0.000043376,0.00005428742,0.9028388,0.009908748,0.01094215],"study_design_scores_gemma":[0.00001682561,0.000008775069,0.0007668342,0.001102526,0.00001820992,0.0001781694,0.1011954,0.000106928,0.0001674956,0.6204387,0.2759752,0.00002492866],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.03990066,0.03635264,0.01521651,0.4853668,0.003877444,0.0001028642,0.0004948311,0.0001354777,0.4185528],"genre_scores_gemma":[0.9429261,0.0136237,0.005788289,0.01709524,0.001780724,0.0002022189,0.0002625018,0.0003020153,0.01801926],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02796738,"threshold_uncertainty_score":0.07466459,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2028141008","doi":"10.1111/j.1740-9713.2012.00583.x","title":"Big Data and City Living – What can it do for Us?","year":2012,"lang":"en","type":"article","venue":"Significance","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":52,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Big data; Data science; Computer science; Gerontology; Medicine; Data mining","authors":[{"name":"Sallie Keller‐McNulty","is_ca":true},{"name":"S. E. Koonin","is_ca":false},{"name":"Stephanie Shipp","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1951561860669645,"gpt":0.4091075099082735,"spread":0.213951323841309,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01785759,0.001261083,0.001519896,0.00377958,0.005698165,0.01850397,0.001992641,0.005512214,0.01440927],"category_scores_gemma":[0.05300175,0.0004961116,0.0009846222,0.00660659,0.01227426,0.03398896,0.00609786,0.01027437,0.005558732],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004796874,"about_ca_system_score_gemma":0.009342521,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01410602,"about_ca_topic_score_gemma":0.01818165,"domain_scores_codex":[0.9930764,0.004125853,0.0002179639,0.0004894927,0.001583134,0.0005071593],"domain_scores_gemma":[0.9593498,0.02164808,0.001603476,0.002434691,0.009376304,0.005587614],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001385502,0.00007796159,0.006786262,0.001071222,0.0001451503,0.00009201043,0.002137464,0.000525242,0.0001100163,0.08505011,0.7742192,0.1296468],"study_design_scores_gemma":[0.00003425351,0.00005618418,0.003716131,0.003623077,0.00006510055,0.00009401671,0.0147327,0.0008827598,0.0002876674,0.4138517,0.5625432,0.0001131649],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.0009706038,0.03437746,0.001274592,0.9524701,0.004936507,0.00001235371,0.0002502555,0.00007326006,0.005634932],"genre_scores_gemma":[0.1680884,0.2939435,0.01427882,0.4613808,0.0348965,0.0003097014,0.001605841,0.0007374379,0.02475909],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.01850397,"threshold_uncertainty_score":0.09444106,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2076133938","doi":"10.7202/013259ar","title":"Differences in News Translation between Broadcasting and Newspapers: A Case Study of Korean-English Translation1","year":2006,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":50,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Newspaper; Linguistics; Translation (biology); Rendering (computer graphics); Broadcasting (networking); Literary translation; Computer science; Advertising; History; Media studies; Sociology; Artificial intelligence; Business","authors":[{"name":"Changsoo Lee","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1200677126363066,"gpt":0.348027203817123,"spread":0.2279594911808164,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008823019,0.0004700961,0.0004338752,0.001207952,0.004842316,0.003341392,0.001037553,0.001372427,0.002964046],"category_scores_gemma":[0.0341279,0.0004162601,0.0003410451,0.00378758,0.002896169,0.00256628,0.001854799,0.001565061,0.0005436064],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002984254,"about_ca_system_score_gemma":0.001920604,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01223083,"about_ca_topic_score_gemma":0.02467299,"domain_scores_codex":[0.9886026,0.009022247,0.0005557744,0.0004219445,0.0008187041,0.0005788244],"domain_scores_gemma":[0.9609911,0.0310138,0.002721425,0.00163386,0.002696353,0.0009433843],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0003037269,0.0004431156,0.02579662,0.0003134692,0.00002076646,0.01339914,0.9288678,0.0002247614,0.002527438,0.002637861,0.0009374424,0.0245279],"study_design_scores_gemma":[0.00008932937,0.0003994955,0.03409625,0.0001338195,0.00006046921,0.005066054,0.9358729,0.00105558,0.00369828,0.0008936307,0.0185783,0.00005594515],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9933906,0.0001622297,0.0006909362,0.0004725259,0.0000176555,0.00006133025,0.00003948172,0.00001025978,0.005155011],"genre_scores_gemma":[0.9960595,0.000260309,0.001590939,0.000136286,0.00001699927,0.00004116597,0.00005953485,0.00003606573,0.001799297],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01223083,"threshold_uncertainty_score":0.04666114,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4395027451","doi":"10.1109/comst.2024.3392642","title":"The Metaverse: Survey, Trends, Novel Pipeline Ecosystem &amp; Future Directions","year":2024,"lang":"en","type":"article","venue":"IEEE Communications Surveys & Tutorials","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":48,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"École de Technologie Supérieure; Concordia University; Polytechnique Montréal","funders":"","keywords":"Pipeline (software); Ecosystem; Environmental resource management; Computer science; Environmental science; Data science; Ecology; Biology","authors":[{"name":"Hani Sami","is_ca":false},{"name":"Ahmad Hammoud","is_ca":false},{"name":"Mohamad Arafeh","is_ca":false},{"name":"Mohamad Wazzeh","is_ca":false},{"name":"Sarhad Arisdakessian","is_ca":false},{"name":"Mario Chahoud","is_ca":false},{"name":"Osama Wehbi","is_ca":false},{"name":"M. Ajaj","is_ca":false},{"name":"Azzam Mourad","is_ca":false},{"name":"Hadi Otrok","is_ca":false},{"name":"Omar Abdel Wahab","is_ca":true},{"name":"Rabeb Mizouni","is_ca":false},{"name":"Jamal Bentahar","is_ca":true},{"name":"Chamseddine Talhi","is_ca":true},{"name":"Zbigniew Dziong","is_ca":true},{"name":"Ernesto Damiani","is_ca":false},{"name":"Mohsen Guizani","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1336653216643221,"gpt":0.4212711845858061,"spread":0.287605862921484,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01228736,0.0007940393,0.0007491508,0.01058539,0.001871508,0.0116632,0.002259714,0.002798535,0.01256547],"category_scores_gemma":[0.02102436,0.0007622154,0.001219451,0.01424761,0.002741883,0.0307233,0.005565188,0.002943034,0.004052611],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003225456,"about_ca_system_score_gemma":0.01193099,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003434593,"about_ca_topic_score_gemma":0.005049976,"domain_scores_codex":[0.9951587,0.001382964,0.0006395439,0.0005872258,0.001593783,0.0006376961],"domain_scores_gemma":[0.9754065,0.01242089,0.002298588,0.001177649,0.006404391,0.002292026],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001162745,0.0001571877,0.007186787,0.01502629,0.00004754697,0.0001994123,0.003253033,0.0005867315,0.001236496,0.09971292,0.06479354,0.8076837],"study_design_scores_gemma":[0.000007131546,0.0001187807,0.002376654,0.008939995,0.00006403645,0.0006331875,0.004677708,0.0005972672,0.0007670869,0.01269478,0.9690745,0.00004880988],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.01520511,0.8371115,0.02766234,0.05300137,0.00305889,0.0002922956,0.001471519,0.001262766,0.06093429],"genre_scores_gemma":[0.08928753,0.83538,0.04130668,0.01024934,0.001992662,0.0004112436,0.003262341,0.0007593178,0.0173509],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.01256547,"threshold_uncertainty_score":0.06498259,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3168278108","doi":"10.1017/pan.2021.37","title":"Cross-Domain Topic Classification for Political Texts","year":2021,"lang":"en","type":"article","venue":"Political Analysis","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":47,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"H2020 European Research Council; University of Essex; Università Bocconi; Eidgenössische Technische Hochschule Zürich; Wissenschaftszentrum Berlin für Sozialforschung; York University; London School of Economics and Political Science","keywords":"Computer science; Classifier (UML); Domain (mathematical analysis); Artificial intelligence; Natural language processing; Labeled data; Machine learning; Mathematics","authors":[{"name":"Moritz Osnabrügge","is_ca":false},{"name":"Elliott Ash","is_ca":false},{"name":"Massimo Morelli","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08141688647427724,"gpt":0.4642052680106769,"spread":0.3827883815363996,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01609405,0.001136573,0.001105122,0.008384028,0.001634625,0.003365829,0.001240072,0.001686829,0.002248029],"category_scores_gemma":[0.03532381,0.0003638859,0.001251805,0.004605992,0.001123123,0.003000078,0.002468018,0.002351138,0.001611406],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001465748,"about_ca_system_score_gemma":0.0009550177,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002408859,"about_ca_topic_score_gemma":0.002408343,"domain_scores_codex":[0.9903688,0.005652858,0.0005649994,0.001836879,0.001174248,0.0004021817],"domain_scores_gemma":[0.9542038,0.0344121,0.002274071,0.003971849,0.004387747,0.0007503316],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00180432,0.001544241,0.08518912,0.0009282091,0.0008309933,0.0005330787,0.00323352,0.1425156,0.01454544,0.01548045,0.01677325,0.7166218],"study_design_scores_gemma":[0.00005132523,0.0001412428,0.01538309,0.00006976205,0.00006134583,0.0002076156,0.0008662455,0.9575075,0.007924411,0.01161626,0.006119051,0.00005215597],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3617499,0.002473823,0.6189905,0.0007665396,0.0003785077,0.0006810228,0.002635963,0.003460733,0.008863056],"genre_scores_gemma":[0.8198316,0.0003105484,0.1701942,0.0001131599,0.000236993,0.0004322126,0.006234217,0.0002452258,0.002401843],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01609405,"threshold_uncertainty_score":0.08511448,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3170686427","doi":"10.3389/frai.2021.664737","title":"Gender Bias in the News: A Scalable Topic Modelling and Visualization Framework","year":2021,"lang":"en","type":"article","venue":"Frontiers in Artificial Intelligence","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":47,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true},"ca_institutions":"Simon Fraser University","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Simon Fraser University","keywords":"Latent Dirichlet allocation; Topic model; Mainstream; Representation (politics); Entertainment; Politics; Computer science; Data science; Psychology; Political science; Artificial intelligence","authors":[{"name":"P. Prabhakar Rao","is_ca":true},{"name":"Maite Taboada","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1975284295315101,"gpt":0.4108134902179633,"spread":0.2132850606864532,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005702454,0.001201477,0.0009848775,0.006737544,0.0009659137,0.003983969,0.00181723,0.001373507,0.004086104],"category_scores_gemma":[0.02157813,0.0006851092,0.00174609,0.004828345,0.0005811678,0.003356377,0.002645893,0.001748901,0.001992234],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001069979,"about_ca_system_score_gemma":0.001579859,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0078665,"about_ca_topic_score_gemma":0.01079803,"domain_scores_codex":[0.997705,0.000899907,0.000176151,0.0005540252,0.0005173762,0.0001475867],"domain_scores_gemma":[0.9915964,0.005048982,0.0007771961,0.0009103052,0.001373392,0.0002938274],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009333238,0.0003307243,0.02362659,0.0008101967,0.0004129314,0.0005459887,0.007223209,0.0557167,0.01630842,0.0301774,0.05056689,0.8133477],"study_design_scores_gemma":[0.0001173254,0.00009749133,0.006450003,0.0001304195,0.000116844,0.0002389995,0.001105406,0.8795354,0.006180704,0.06373552,0.04220358,0.00008814211],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01966578,0.0009736079,0.9551451,0.001100165,0.0001779229,0.0002940831,0.005778713,0.01404046,0.002824096],"genre_scores_gemma":[0.2203066,0.000686797,0.7656395,0.0002144107,0.000358457,0.000918571,0.007809246,0.001235464,0.002830961],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0078665,"threshold_uncertainty_score":0.0301578,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4292819996","doi":"10.1177/14614448221113923","title":"Participatory censorship: How online fandom community facilitates authoritarian rule","year":2022,"lang":"en","type":"article","venue":"New Media & Society","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":46,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University; Concordia University","funders":"","keywords":"Censorship; Authoritarianism; Social media; Sociology; Citizen journalism; Politics; Online community; Public relations; Political science; Media studies; Democracy; Law","authors":[{"name":"Zhifan Luo","is_ca":true},{"name":"Muyang Li","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1814214380344424,"gpt":0.3824314028423084,"spread":0.2010099648078661,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002252741,0.0003015562,0.0002334591,0.001023257,0.003145824,0.003961659,0.0006658819,0.000584429,0.006132469],"category_scores_gemma":[0.01059469,0.0001742123,0.0002646635,0.0005825668,0.003584065,0.00356434,0.002953172,0.0008388403,0.0005480364],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006848514,"about_ca_system_score_gemma":0.001026049,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00251757,"about_ca_topic_score_gemma":0.003982402,"domain_scores_codex":[0.9981343,0.001082586,0.00005310239,0.0002644985,0.000238811,0.0002267845],"domain_scores_gemma":[0.9927757,0.003923969,0.001177182,0.0008053681,0.0004928585,0.00082496],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0005068778,0.0005833894,0.1859696,0.0004530936,0.00008247302,0.002642686,0.524986,0.001264344,0.01255403,0.0960454,0.006509014,0.1684032],"study_design_scores_gemma":[0.0001854552,0.0007286282,0.244816,0.0005222863,0.0002036628,0.002104531,0.4165355,0.02664468,0.009352216,0.1243866,0.174331,0.0001893743],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9296966,0.0001995594,0.01081642,0.000875465,0.00006969808,0.0000786985,0.00005298407,0.0001243912,0.05808625],"genre_scores_gemma":[0.9970126,0.00003568735,0.000798957,0.00004035077,0.00001334339,0.00001466143,0.00002171403,0.00001539265,0.002047196],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006132469,"threshold_uncertainty_score":0.02051508,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4398440537","doi":"10.7910/dvn/dus88v","title":"2019 Canadian Election Study (CES) - Online Survey","year":2020,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":40,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Toronto; Toronto Metropolitan University; Université du Québec à Montréal; Western University","funders":"","keywords":"Geography","authors":[{"name":"Laura B. Stephenson","is_ca":true},{"name":"Allison Harell","is_ca":true},{"name":"Daniel Rubenson","is_ca":true},{"name":"Peter John Loewen","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06690661034692758,"gpt":0.365365284635189,"spread":0.2984586742882614,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001913449,0.001313912,0.001154839,0.003984853,0.003034495,0.002189216,0.003035595,0.001109008,0.03812497],"category_scores_gemma":[0.01098672,0.0005603258,0.0008057718,0.009186035,0.0004455158,0.0007251388,0.001947309,0.001804513,0.02519728],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01240865,"about_ca_system_score_gemma":0.02805713,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.9031727,"about_ca_topic_score_gemma":0.9582323,"domain_scores_codex":[0.9981185,0.000215187,0.0001902068,0.000319185,0.0007667373,0.0003900843],"domain_scores_gemma":[0.9923521,0.0005420208,0.0003920514,0.0006875601,0.005249469,0.0007768366],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00006107078,0.00002052932,0.002577793,0.0001546131,0.00002305428,0.00001908418,0.0000419833,0.0000750957,0.00003066641,0.0004099269,0.9945759,0.00201031],"study_design_scores_gemma":[0.0003348954,0.00002432039,0.0716226,0.0003817103,0.00007079558,0.00005790214,0.0004129546,0.0005791587,0.0002940115,0.0007367489,0.9253902,0.00009466722],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0004848986,0.00003162421,0.00004542843,0.00008499785,0.00001977765,0.00005118862,0.9979443,0.00007738271,0.001260429],"genre_scores_gemma":[0.001431506,0.0000505382,0.0003602276,0.0001059747,0.00001022835,0.0002579744,0.9956278,0.00004101578,0.002114723],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.09682727,"threshold_uncertainty_score":0.194795,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2810369529","doi":"10.1017/s1049096518000926","title":"Data Access, Transparency, and Replication: New Insights from the Political Behavior Literature","year":2018,"lang":"en","type":"article","venue":"PS Political Science & Politics","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":40,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Transparency (behavior); Politics; Replication (statistics); Political science; Data sharing; Public relations; Law; Statistics; Medicine; Mathematics","authors":[{"name":"Daniel Stockemer","is_ca":true},{"name":"Sebastian Koehler","is_ca":false},{"name":"Tobias Lentz","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1624169664032016,"gpt":0.4777498636635403,"spread":0.3153328972603388,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4539078,0.001243743,0.002732375,0.01667118,0.006775951,0.01997571,0.006566374,0.006356574,0.007763606],"category_scores_gemma":[0.8401303,0.001886123,0.003180887,0.02611863,0.03322706,0.03989817,0.01151196,0.008426318,0.0007067611],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006287876,"about_ca_system_score_gemma":0.01204917,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005947894,"about_ca_topic_score_gemma":0.004792255,"domain_scores_codex":[0.3914568,0.4978106,0.03183162,0.02596979,0.04852914,0.004402121],"domain_scores_gemma":[0.0278797,0.88115,0.03036334,0.04601805,0.01369189,0.0008970532],"domain_codex":"methods","domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.001121016,0.0004749243,0.1478918,0.007417325,0.002344832,0.003678004,0.1335853,0.005225219,0.001132101,0.4801008,0.01522645,0.2018023],"study_design_scores_gemma":[0.0002221815,0.000218539,0.02944717,0.004812046,0.001069192,0.001648767,0.03155587,0.02960508,0.001641046,0.8664362,0.03303528,0.0003086666],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.23653,0.03296794,0.5109562,0.1653443,0.002214056,0.001909493,0.003309848,0.0004902197,0.04627799],"genre_scores_gemma":[0.9374876,0.002595874,0.05312662,0.003102863,0.001327703,0.001035725,0.000515434,0.0001514248,0.0006568356],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9833288,"threshold_uncertainty_score":0.673429,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2886358838","doi":"10.18653/v1/w16-56","title":"Proceedings of the First Workshop on NLP and Computational Social Science","year":2016,"lang":"en","type":"paratext","venue":"","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":35,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Pacific Northwest National Laboratory; Office of Naval Research; University of Toronto; University of California, San Diego; Defense Advanced Research Projects Agency; Universiteit van Tilburg; University of Massachusetts Amherst; Freie Universität Berlin; Centre National de la Recherche Scientifique; University of Melbourne; Institute for Quantitative Social Science, Harvard University; University of Pittsburgh; Korea Advanced Institute of Science and Technology; University of Minnesota; National Science Foundation; Princeton University; University of Washington; Harvard Business School; Ohio State University; University of Pennsylvania; Multidisciplinary University Research Initiative; McGill University; Microsoft Research; University College London; Carnegie Mellon University","keywords":"Computer science; Artificial intelligence; Natural language processing; Data science","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.04071723295587542,"gpt":0.3739174452502604,"spread":0.333200212294385,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02199683,0.00120467,0.001483971,0.003436028,0.002884985,0.01171427,0.002565999,0.00276514,0.05911018],"category_scores_gemma":[0.04360328,0.0006076008,0.002117282,0.003899059,0.003407496,0.01476903,0.009023434,0.008818437,0.01965215],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004345092,"about_ca_system_score_gemma":0.009424166,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00647386,"about_ca_topic_score_gemma":0.01038017,"domain_scores_codex":[0.9903894,0.006111412,0.0005280013,0.001005246,0.001634184,0.0003317586],"domain_scores_gemma":[0.9594934,0.02541613,0.0004489645,0.005187103,0.005504722,0.003949746],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001937998,0.0001557224,0.0005497385,0.0006041529,0.00007166136,0.0002232425,0.001592659,0.002756229,0.0009933235,0.09751593,0.7138554,0.1814882],"study_design_scores_gemma":[0.00002602139,0.00002117778,0.0004145324,0.0002764121,0.00001769556,0.00011566,0.0004948407,0.005255272,0.001085262,0.07591972,0.9163435,0.00002987891],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.01041044,0.01777423,0.5279294,0.1684219,0.05334179,0.0009978532,0.01825934,0.006002965,0.1968622],"genre_scores_gemma":[0.0697201,0.01993094,0.3054913,0.01442774,0.01590857,0.002501034,0.05108863,0.009120409,0.5118112],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.05911018,"threshold_uncertainty_score":0.1977432,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4310576792","doi":"10.1177/00491241221123088","title":"From Ends to Means: The Promise of Computational Text Analysis for Theoretically Driven Sociological Research","year":2022,"lang":"en","type":"article","venue":"Sociological Methods & Research","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":34,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Skepticism; Field (mathematics); Relevance (law); Management science; Set (abstract data type); Computational model; Epistemology; Selection (genetic algorithm); Computational sociology; Data science; Process (computing); Sociology; Artificial intelligence; Political science","authors":[{"name":"Bart Bonikowski","is_ca":false},{"name":"Laura K. Nelson","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4486729254285306,"gpt":0.6277772920025255,"spread":0.1791043665739949,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04772533,0.001330582,0.001562148,0.00736503,0.004788702,0.02512696,0.004063154,0.003994727,0.0109742],"category_scores_gemma":[0.1537814,0.0009345489,0.001510869,0.006246351,0.02453191,0.04640342,0.01169228,0.008966723,0.002558948],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003286366,"about_ca_system_score_gemma":0.005334416,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001173181,"about_ca_topic_score_gemma":0.002003332,"domain_scores_codex":[0.9519697,0.04001203,0.001231462,0.001897996,0.004534476,0.0003542927],"domain_scores_gemma":[0.6421149,0.3278784,0.004716418,0.01919315,0.004580602,0.001516587],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00005955173,0.00004108351,0.001059392,0.000638399,0.00005951426,0.00009275677,0.00741728,0.001577698,0.0003419457,0.9248182,0.009976122,0.05391802],"study_design_scores_gemma":[0.00001133403,0.000009916401,0.0001988179,0.0003372777,0.00001011769,0.00006134162,0.001506593,0.005324844,0.0002654535,0.9622685,0.0299725,0.00003345012],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0111026,0.009592866,0.8501857,0.07957366,0.001622943,0.0003965491,0.0006504717,0.0006972714,0.04617798],"genre_scores_gemma":[0.19834,0.009189635,0.7763157,0.005223824,0.002626496,0.001661489,0.0007865609,0.0008063406,0.005049944],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9522747,"threshold_uncertainty_score":0.2523987,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3113091126","doi":"10.1016/j.bdr.2020.100173","title":"Agreeing to Disagree: Choosing Among Eight Topic-Modeling Methods","year":2020,"lang":"en","type":"article","venue":"Big Data Research","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":31,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University; University of British Columbia","funders":"Hong Kong Polytechnic University","keywords":"Latent Dirichlet allocation; Topic model; Computer science; Data science; Artificial intelligence; Coherence (philosophical gambling strategy); Field (mathematics); Context (archaeology); Machine learning; Management science; Natural language processing","authors":[{"name":"Qiang Fu","is_ca":true},{"name":"Yufan Zhuang","is_ca":false},{"name":"Jiaxin Gu","is_ca":true},{"name":"Yushu Zhu","is_ca":true},{"name":"Xin Guo","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.8645336607084751,"gpt":0.6295130273678069,"spread":0.2350206333406683,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1976716,0.002535491,0.003023694,0.006498765,0.004347725,0.007980894,0.005831443,0.005154657,0.006813993],"category_scores_gemma":[0.3329974,0.001635352,0.004478248,0.004767443,0.003544451,0.01111379,0.007425272,0.007263245,0.002170931],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001750238,"about_ca_system_score_gemma":0.003889166,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00136597,"about_ca_topic_score_gemma":0.004318592,"domain_scores_codex":[0.8347772,0.1381306,0.007954801,0.00594509,0.01091581,0.002276591],"domain_scores_gemma":[0.3774642,0.5812154,0.008933019,0.01187027,0.01562411,0.004893096],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01206268,0.002374437,0.08158297,0.004764759,0.004213657,0.0002374645,0.02465527,0.008747776,0.00364413,0.04218261,0.02420812,0.7913261],"study_design_scores_gemma":[0.006882404,0.002645764,0.04922847,0.004978783,0.006237485,0.0008355011,0.0321158,0.4033126,0.01330726,0.4376624,0.04152385,0.00126977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2081795,0.004004317,0.7457227,0.01238216,0.001131329,0.006634318,0.00131667,0.001910983,0.01871799],"genre_scores_gemma":[0.440204,0.00103177,0.5450583,0.001797486,0.0004251892,0.007083504,0.001521211,0.0005649134,0.002313689],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1976716,"threshold_uncertainty_score":0.9894137,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4391464262","doi":"10.1162/qss_a_00285","title":"Large-scale text analysis using generative language models: A case study in discovering public value expressions in AI patents","year":2024,"lang":"en","type":"article","venue":"Quantitative Science Studies","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":30,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Biotechnology and Biological Sciences Research Council; Directorate for Biological Sciences; Snap","keywords":"Generative grammar; Natural language processing; Scale (ratio); Value (mathematics); Computer science; Generative model; Artificial intelligence; Linguistics; Machine learning; Geography; Cartography; Philosophy","authors":[{"name":"Sergio Pelaez","is_ca":false},{"name":"Gaurav Verma","is_ca":false},{"name":"Bárbara Ribeiro","is_ca":false},{"name":"Philip Shapira","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2553169436416557,"gpt":0.5238114656364624,"spread":0.2684945219948067,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.009732377,0.0004807042,0.0003909678,0.002477261,0.001126441,0.001623267,0.001211728,0.001221504,0.0009783691],"category_scores_gemma":[0.03844003,0.0002860266,0.000910308,0.002155135,0.002070491,0.002982442,0.0012794,0.00193265,0.0003083381],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001554639,"about_ca_system_score_gemma":0.001106229,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005144007,"about_ca_topic_score_gemma":0.007717496,"domain_scores_codex":[0.9923357,0.005856684,0.0002286763,0.0005680568,0.0008840446,0.0001268321],"domain_scores_gemma":[0.8808862,0.110239,0.002641988,0.003659954,0.002168568,0.000404285],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0006728482,0.001293335,0.0704818,0.0008389166,0.0002853562,0.005061483,0.01671035,0.2753342,0.03062206,0.1198804,0.01620342,0.4626159],"study_design_scores_gemma":[0.00005465228,0.0001044254,0.005929768,0.0000473082,0.00003622869,0.0003605183,0.001314008,0.9205689,0.01283388,0.05301091,0.005677835,0.00006151064],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4625821,0.0004235735,0.5261752,0.003677132,0.00004844311,0.0003389884,0.0008757057,0.001982134,0.003896722],"genre_scores_gemma":[0.77308,0.0001086626,0.2247087,0.0002624562,0.00004158982,0.0001624041,0.0007742187,0.0001439468,0.000718209],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9975227,"threshold_uncertainty_score":0.05147034,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2070320835","doi":"10.1002/asi.22797","title":"In their own image? a comparison of doctoral students' and faculty members' referencing behavior","year":2013,"lang":"en","type":"article","venue":"Journal of the American Society for Information Science and Technology","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":29,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal; Université du Québec à Montréal","funders":"","keywords":"Doctoral dissertation; Library science; Psychology; University faculty; Sociology; Medical education; Higher education; Medicine; Political science; Computer science","authors":[{"name":"Vincent Larivière","is_ca":true},{"name":"Cassidy R. Sugimoto","is_ca":false},{"name":"Pierrette Bergeron","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05346329054850051,"gpt":0.4297384152809274,"spread":0.3762751247324269,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.007929335,0.0002023167,0.0003952682,0.003320553,0.001688967,0.003210401,0.0005585313,0.0008804615,0.003866795],"category_scores_gemma":[0.05805666,0.0001848122,0.0003403573,0.002547928,0.0009260988,0.001953552,0.001712295,0.0007338054,0.0009363752],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006489658,"about_ca_system_score_gemma":0.0008654114,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00117579,"about_ca_topic_score_gemma":0.00161652,"domain_scores_codex":[0.9963949,0.001611733,0.0002832178,0.0002551989,0.001217375,0.0002375999],"domain_scores_gemma":[0.9616044,0.01997167,0.008865598,0.002119964,0.003993724,0.003444592],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003344366,0.0002957562,0.5115168,0.0003647825,0.0001640563,0.0006323202,0.2559977,0.0001272021,0.006922673,0.003107872,0.004471304,0.2160652],"study_design_scores_gemma":[0.00003549302,0.0005848764,0.6504341,0.000380898,0.0001320828,0.002508675,0.2938052,0.0008386973,0.003844169,0.003290178,0.04403163,0.0001140389],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9912025,0.001140671,0.0005809635,0.0008030339,0.0000424998,0.0000169224,0.0000643086,0.00002011086,0.006128945],"genre_scores_gemma":[0.9953665,0.0008566379,0.0005970088,0.0002488979,0.00006712043,0.00001536593,0.00009263422,0.00001583378,0.002740013],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9966794,"threshold_uncertainty_score":0.04193485,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2077014275","doi":"10.3138/cjccj.2013.e44","title":"Beyond Frequency: Perceived Realism and the <i>CSI</i> Effect","year":2014,"lang":"en","type":"article","venue":"Canadian Journal of Criminology and Criminal Justice/La Revue canadienne de criminologie et de justice pénale","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":29,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"Carleton University","funders":"","keywords":"Realism; Context (archaeology); Psychology; Social psychology; Moderation; Criminal justice; Economic Justice; Criminology; Political science; Law","authors":[{"name":"Evelyn M. Maeder","is_ca":true},{"name":"Richard Corbett","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05552850342321045,"gpt":0.3268272035254285,"spread":0.271298700102218,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005912798,0.0003577934,0.0004072414,0.0009425733,0.0006235309,0.002211119,0.0005884952,0.0009089654,0.008657443],"category_scores_gemma":[0.04933463,0.0003238487,0.0007673768,0.0006408396,0.001794104,0.001417973,0.001592111,0.001936778,0.0003652598],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000722616,"about_ca_system_score_gemma":0.0004141052,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001994388,"about_ca_topic_score_gemma":0.001995339,"domain_scores_codex":[0.9951785,0.002785763,0.0002949544,0.0004470883,0.001066168,0.0002275862],"domain_scores_gemma":[0.8677313,0.1082882,0.01367506,0.00515109,0.002794637,0.002359729],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.004865502,0.001579318,0.8676116,0.001163304,0.001333686,0.000506759,0.0200607,0.002162972,0.0233895,0.01903209,0.001638986,0.05665555],"study_design_scores_gemma":[0.0001391754,0.0009677597,0.9839401,0.0001970711,0.0003224911,0.0002935024,0.003174475,0.001738402,0.001833449,0.004572206,0.002752076,0.0000693184],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.944891,0.0004955663,0.005421668,0.0009720579,0.00007771451,0.0001688691,0.0002146113,0.00006910649,0.04768948],"genre_scores_gemma":[0.9973657,0.0001102195,0.001310425,0.0001984946,0.00005511759,0.00004959601,0.00007852398,0.00002007146,0.0008119291],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9980056,"threshold_uncertainty_score":0.03127021,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4410554739","doi":"10.1177/00491241251339188","title":"Updating “The Future of Coding”: Qualitative Coding with Generative Large Language Models","year":2025,"lang":"en","type":"article","venue":"Sociological Methods & Research","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":27,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"Russell Sage Foundation","keywords":"Coding (social sciences); Generative grammar; Computer science; Natural language processing; Linguistics; Qualitative research; Artificial intelligence; Sociology; Social science","authors":[{"name":"Nga Than","is_ca":false},{"name":"Leanne Fan","is_ca":false},{"name":"Tina Law","is_ca":false},{"name":"Laura K. Nelson","is_ca":true},{"name":"Leslie McCall","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2818745422435381,"gpt":0.6225325648240265,"spread":0.3406580225804884,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1450383,0.001423208,0.0009625413,0.004098671,0.004843164,0.01218582,0.004987217,0.002798446,0.009748672],"category_scores_gemma":[0.4446427,0.001556783,0.001525877,0.004195776,0.02694574,0.02438786,0.01067899,0.006763448,0.002623544],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01021281,"about_ca_system_score_gemma":0.01473867,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006793946,"about_ca_topic_score_gemma":0.008921273,"domain_scores_codex":[0.7985219,0.1782885,0.004230367,0.007846711,0.0100183,0.001094163],"domain_scores_gemma":[0.4858758,0.3883091,0.01418253,0.07850019,0.03091722,0.002215277],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001875412,0.00007070487,0.004540558,0.00127844,0.00009679854,0.0001957302,0.13612,0.006370578,0.001849917,0.7195998,0.01441154,0.1152783],"study_design_scores_gemma":[0.00007219223,0.00003979199,0.0008097984,0.001592215,0.00003738846,0.0001752657,0.01866881,0.03233928,0.002946022,0.8712533,0.07190628,0.0001596892],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01357034,0.000486505,0.9591738,0.01505915,0.0005261713,0.0005324371,0.0007495381,0.0008488533,0.009053214],"genre_scores_gemma":[0.260394,0.0005273927,0.7280648,0.003789688,0.0001945691,0.002735575,0.0007785886,0.001122387,0.002392892],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8549617,"threshold_uncertainty_score":0.767045,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2800743636","doi":"10.2139/ssrn.3067734","title":"The R Package sentometrics to Compute, Aggregate and Predict with Textual Sentiment","year":2017,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":25,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Sherbrooke; Center for Interuniversity Research and Analysis on Organizations; HEC Montréal","funders":"","keywords":"Indexation; Computer science; Sentiment analysis; Aggregate (composite); R package; Workflow; Index (typography); Information retrieval; Natural language processing; Database; World Wide Web; Programming language; Economics; Monetary policy","authors":[{"name":"David Ardia","is_ca":true},{"name":"Keven Bluteau","is_ca":true},{"name":"Samuel Borms","is_ca":false},{"name":"Kris Boudt","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01744146395906499,"gpt":0.334739107158963,"spread":0.3172976431998981,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002460458,0.002330006,0.001323229,0.004331083,0.000657852,0.0029288,0.00110035,0.0006575393,0.05071777],"category_scores_gemma":[0.03142991,0.0008788715,0.002298909,0.00330414,0.0006096177,0.001799749,0.002402844,0.001441217,0.05857712],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000650597,"about_ca_system_score_gemma":0.002195629,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002717016,"about_ca_topic_score_gemma":0.005643324,"domain_scores_codex":[0.9982594,0.000652709,0.0001670931,0.0003822597,0.0003997426,0.0001388509],"domain_scores_gemma":[0.9932184,0.003719028,0.0007084141,0.001404353,0.0007685319,0.0001812889],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005598637,0.0001829041,0.01764086,0.002208114,0.001654911,0.0003998728,0.0008408558,0.01151215,0.005129963,0.02948746,0.6797498,0.2506333],"study_design_scores_gemma":[0.0005437144,0.0003123517,0.01973661,0.0004044075,0.0009470693,0.001170824,0.0004234217,0.1766257,0.01763245,0.1097301,0.672181,0.0002923587],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.02042582,0.0008061404,0.4276754,0.001189023,0.0009845167,0.0006371311,0.167059,0.3633435,0.01787943],"genre_scores_gemma":[0.1162724,0.0009508677,0.5937206,0.0008399261,0.0006456923,0.004023226,0.1158788,0.1301584,0.03751003],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.05071777,"threshold_uncertainty_score":0.1696678,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2478027989","doi":"10.1177/0049124116661575","title":"A Novel Sequential Mixed-method Technique for Contrastive Analysis of Unscripted Qualitative Data","year":2016,"lang":"en","type":"article","venue":"Sociological Methods & Research","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":25,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"NeuroDevNet; University of British Columbia","funders":"","keywords":"Computer science; Multimethodology; Qualitative research; Qualitative property; Subject (documents); Data science; Contrastive analysis; Psychology; Linguistics; Mathematics education; Machine learning; Sociology; World Wide Web; Social science","authors":[{"name":"Laura Y. Cabrera","is_ca":true},{"name":"Peter B. Reiner","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.6465089207335274,"gpt":0.6967122529922157,"spread":0.05020333225868834,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1107211,0.003844063,0.003013406,0.00770756,0.004106853,0.005726018,0.005496664,0.002223079,0.0242974],"category_scores_gemma":[0.2458278,0.002261039,0.003492907,0.008246874,0.00477624,0.003196929,0.006948769,0.004993264,0.00370356],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004332592,"about_ca_system_score_gemma":0.007304009,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00249213,"about_ca_topic_score_gemma":0.004752346,"domain_scores_codex":[0.8098165,0.1514556,0.007825671,0.01659565,0.01314559,0.001161047],"domain_scores_gemma":[0.6839737,0.2463913,0.01374134,0.03053029,0.02414994,0.001213505],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002671395,0.001230205,0.003924994,0.008370697,0.001672626,0.000590719,0.04234854,0.008973056,0.02556659,0.1666125,0.01472953,0.7233091],"study_design_scores_gemma":[0.004818253,0.005515394,0.01189842,0.00367811,0.001930438,0.000985616,0.01132489,0.2048148,0.06432299,0.4077209,0.2817275,0.00126266],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002956113,0.00006445442,0.9790315,0.0001399584,0.0002093292,0.01422183,0.0005015,0.0008289141,0.002046479],"genre_scores_gemma":[0.007341634,0.00002810428,0.9423391,0.0000957362,0.00003720734,0.0493578,0.0001315364,0.0002113949,0.0004575312],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8892788,"threshold_uncertainty_score":0.5855564,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4213310220","doi":"10.1016/j.jbi.2022.104034","title":"Non-negative matrix factorization temporal topic models and clinical text data identify COVID-19 pandemic effects on primary healthcare and community health in Toronto, Canada","year":2022,"lang":"en","type":"article","venue":"Journal of Biomedical Informatics","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":25,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true},"ca_institutions":"North York General Hospital; Institute for Clinical Evaluative Sciences; Public Health Ontario; University of Toronto","funders":"Canadian Institutes of Health Research","keywords":"Pandemic; Health care; Computer science; Medicine; Artificial intelligence; Psychology; Coronavirus disease 2019 (COVID-19); Disease; Infectious disease (medical specialty); Pathology","authors":[{"name":"Christopher Meaney","is_ca":true},{"name":"Michael Escobar","is_ca":true},{"name":"Rahim Moineddin","is_ca":true},{"name":"Thérèse A. Stukel","is_ca":true},{"name":"Sumeet Kalia","is_ca":true},{"name":"Babak Aliarzadeh","is_ca":true},{"name":"Tao Chen","is_ca":true},{"name":"Braden O’Neill","is_ca":true},{"name":"Michelle Greiver","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1456593808156004,"gpt":0.4946154215590821,"spread":0.3489560407434817,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002253958,0.0004588792,0.0004963715,0.002049413,0.001491893,0.001663075,0.001175751,0.0008050067,0.002468402],"category_scores_gemma":[0.01112576,0.0003390745,0.0007038919,0.003480069,0.0005754854,0.0007063507,0.001006661,0.001319419,0.0003915497],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01568915,"about_ca_system_score_gemma":0.02584252,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.9863641,"about_ca_topic_score_gemma":0.9885566,"domain_scores_codex":[0.9989863,0.0002933113,0.00008248969,0.000205368,0.000144717,0.0002878066],"domain_scores_gemma":[0.9929897,0.003140101,0.0006776635,0.0002518886,0.002173845,0.0007668616],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0006361657,0.0002039025,0.8970774,0.0003425514,0.0004458089,0.0007242951,0.003953154,0.01736797,0.000688289,0.002713934,0.04312843,0.03271809],"study_design_scores_gemma":[0.00005879752,0.00005402185,0.8795549,0.0001837939,0.0002341293,0.0001443608,0.008456063,0.1003827,0.0003303006,0.001644953,0.008882714,0.00007327987],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9407712,0.002326897,0.00393715,0.008020263,0.0001656748,0.0001302703,0.04242449,0.0001223705,0.002101673],"genre_scores_gemma":[0.981106,0.0007353579,0.00161295,0.0002114211,0.00007086399,0.00005451632,0.01352017,0.00002723404,0.002661466],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01568915,"threshold_uncertainty_score":0.1138333,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4317910427","doi":"10.1177/20539517221149106","title":"Formally comparing topic models and human-generated qualitative coding of physician mothers’ experiences of workplace discrimination","year":2023,"lang":"en","type":"article","venue":"Big Data & Society","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":24,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"National Cancer Institute; National Center for Advancing Translational Sciences; National Institute of Arthritis and Musculoskeletal and Skin Diseases; National Human Genome Research Institute","keywords":"Coding (social sciences); Computer science; Thematic analysis; Leverage (statistics); Data science; Qualitative research; Topic model; Context (archaeology); Artificial intelligence; Sociology","authors":[{"name":"Adam S. Miner","is_ca":false},{"name":"Sheridan A. Stewart","is_ca":false},{"name":"Meghan C. Halley","is_ca":false},{"name":"Laura K. Nelson","is_ca":true},{"name":"Eleni Linos","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4102039748049899,"gpt":0.469404413208051,"spread":0.05920043840306116,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1116912,0.0007794771,0.0006169264,0.002912056,0.00154069,0.006772721,0.003183432,0.001750388,0.004091344],"category_scores_gemma":[0.4052806,0.0007219624,0.00147213,0.002903856,0.006000617,0.006448964,0.004291059,0.002529687,0.0005715975],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006786175,"about_ca_system_score_gemma":0.00589964,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005503365,"about_ca_topic_score_gemma":0.006189561,"domain_scores_codex":[0.8477873,0.1369834,0.003519475,0.004960452,0.006008444,0.0007409418],"domain_scores_gemma":[0.2967157,0.6607538,0.01414245,0.01964381,0.007960765,0.0007833885],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001204097,0.0006768336,0.06607608,0.00411088,0.0007068086,0.0004262227,0.1797377,0.1392483,0.003802149,0.4404224,0.007969051,0.1556196],"study_design_scores_gemma":[0.000408226,0.0003648929,0.01671957,0.001700012,0.0001576551,0.0002783891,0.03754101,0.5168979,0.004669337,0.3981003,0.02288653,0.0002760807],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1155023,0.0002540163,0.8711876,0.00298041,0.0001305176,0.001898286,0.001539764,0.0003900952,0.006117],"genre_scores_gemma":[0.5983464,0.0001295785,0.3927039,0.0005076421,0.00005972419,0.005834904,0.001574032,0.0001562036,0.0006877193],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8883088,"threshold_uncertainty_score":0.5906863,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2788605280","doi":"10.63317/37krdg6qf9fs","title":"‘Aye’ or ‘No’? Speech-level Sentiment Analysis of Hansard UK Parliamentary Debate Transcripts","year":2018,"lang":"en","type":"article","venue":"","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":23,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Open Text (Canada)","funders":"","keywords":"Sentiment analysis; Speech recognition; Computer science; Political science; Natural language processing","authors":[{"name":"Gavin Abercrombie","is_ca":false},{"name":"Riza Batista-Navarro","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1028646903379803,"gpt":0.3909686534896928,"spread":0.2881039631517125,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001588528,0.0002711311,0.0002688689,0.00118528,0.0005880399,0.0008821121,0.0002692701,0.0005793381,0.002338622],"category_scores_gemma":[0.01115902,0.000121328,0.0002365552,0.001153873,0.0005354229,0.0003979876,0.0006810623,0.0005624643,0.001456058],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006902806,"about_ca_system_score_gemma":0.0003208215,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004084945,"about_ca_topic_score_gemma":0.008168837,"domain_scores_codex":[0.9979376,0.001023095,0.0001582763,0.0002118048,0.0005159656,0.0001533961],"domain_scores_gemma":[0.9908926,0.006765415,0.0007563072,0.0002628586,0.001146947,0.0001756991],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.005276444,0.0004810185,0.1309262,0.002797529,0.0002344933,0.00448738,0.0822584,0.01483642,0.2166558,0.009249931,0.04595278,0.4868437],"study_design_scores_gemma":[0.000170145,0.0009160556,0.5593746,0.0006260065,0.0001432888,0.00218612,0.0466756,0.1490142,0.1259025,0.00609456,0.1085987,0.0002982896],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.9841983,0.0001886586,0.005670154,0.0003779231,0.00009747879,0.000128752,0.004088611,0.0001431368,0.005106881],"genre_scores_gemma":[0.9827501,0.0001032803,0.006628203,0.00004768175,0.00005602755,0.0001526468,0.006104932,0.0000444236,0.004112816],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.004084945,"threshold_uncertainty_score":0.008401036,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3215074405","doi":"10.1145/3492844","title":"The Computational Thematic Analysis Toolkit","year":2022,"lang":"en","type":"article","venue":"Proceedings of the ACM on Human-Computer Interaction","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":23,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; University of Waterloo","keywords":"Computer science; Transparency (behavior); Thematic analysis; Data science; Variety (cybernetics); Interface (matter); Coding (social sciences); Computational model; Thematic map; Modular design; Computational sociology; Process (computing); Human–computer interaction; World Wide Web; Qualitative research; Artificial intelligence; Sociology","authors":[{"name":"Robert P. Gauthier","is_ca":true},{"name":"James R. Wallace","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07997970442216792,"gpt":0.4028001203780254,"spread":0.3228204159558574,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02817894,0.0020999,0.001642867,0.007563906,0.003409159,0.008543228,0.005618544,0.001712032,0.04851327],"category_scores_gemma":[0.06758262,0.002322917,0.005959668,0.007632795,0.002185082,0.007162398,0.01187579,0.005581058,0.0168544],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003010273,"about_ca_system_score_gemma":0.01181008,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01015654,"about_ca_topic_score_gemma":0.02172053,"domain_scores_codex":[0.9834222,0.009647717,0.002354379,0.001718728,0.002440551,0.0004164003],"domain_scores_gemma":[0.940681,0.04212463,0.002613026,0.0071568,0.006381379,0.001043019],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003950873,0.0001907848,0.002853499,0.008688971,0.0007805998,0.0009770618,0.02423591,0.01209868,0.005660468,0.2141716,0.4026909,0.3272564],"study_design_scores_gemma":[0.0001970468,0.00003552537,0.001311808,0.001342248,0.0001668755,0.0003998484,0.002649216,0.03663697,0.002332212,0.2537362,0.7009918,0.0002002042],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.001948658,0.0005397405,0.899621,0.002142421,0.0004199116,0.002636444,0.04076053,0.03952112,0.01241014],"genre_scores_gemma":[0.007126748,0.0002901435,0.9608316,0.000265869,0.00004524394,0.008475827,0.01433284,0.004830079,0.003801621],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.04851327,"threshold_uncertainty_score":0.162293,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4306178485","doi":"10.3233/jid-220013","title":"Natural language processing (NLP) aided qualitative method in health research","year":2022,"lang":"en","type":"article","venue":"Journal of Integrated Design and Process Science","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":23,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Mount Royal University; Alberta Health Services; Concordia University; University of Calgary","funders":"","keywords":"Computer science; Natural language processing; Sentence; Cluster analysis; Artificial intelligence; Workload; Context (archaeology); Automatic summarization; Qualitative research; Information retrieval; Text mining; Data science; Data mining","authors":[{"name":"Cheligeer Cheligeer","is_ca":true},{"name":"Lin Yang","is_ca":true},{"name":"Tannistha Nandi","is_ca":true},{"name":"Chelsea Doktorchik","is_ca":true},{"name":"Hude Quan","is_ca":true},{"name":"Yong Zeng","is_ca":true},{"name":"Shaminder Singh","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2305358605394815,"gpt":0.5988821724556453,"spread":0.3683463119161638,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1339559,0.001729705,0.001834184,0.007627744,0.00392189,0.006050323,0.004004781,0.001792957,0.02123919],"category_scores_gemma":[0.1744137,0.001565687,0.001615454,0.009767845,0.006934411,0.004672901,0.007253109,0.004084722,0.004616252],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006310608,"about_ca_system_score_gemma":0.0199154,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005084215,"about_ca_topic_score_gemma":0.008799369,"domain_scores_codex":[0.8164564,0.1613568,0.008239961,0.006493032,0.006437418,0.001016366],"domain_scores_gemma":[0.7082444,0.2438751,0.009711579,0.01739353,0.01970423,0.001071198],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001228031,0.000594031,0.003564489,0.02351106,0.0002782898,0.001410244,0.1663933,0.01005433,0.02135552,0.2156343,0.02638942,0.529587],"study_design_scores_gemma":[0.001109282,0.0007313686,0.006233791,0.009975049,0.000282158,0.001024298,0.07001269,0.06577872,0.03129664,0.4301246,0.3826133,0.0008179804],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.007187231,0.0004688476,0.9609296,0.001746621,0.000257083,0.0189776,0.002129784,0.0007674628,0.007535836],"genre_scores_gemma":[0.02134573,0.0002476665,0.9378156,0.00042292,0.00004380371,0.0383817,0.0004089121,0.0001976254,0.001135952],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1339559,"threshold_uncertainty_score":0.7084348,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3034470301","doi":"10.1016/j.forsciint.2020.110364","title":"Social media forensics applied to assessment of post–critical incident social reaction: The case of the 2017 Manchester Arena terrorist attack","year":2020,"lang":"en","type":"article","venue":"Forensic Science International","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":23,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal; Université du Québec à Trois-Rivières","funders":"Economic and Social Research Council","keywords":"Latent Dirichlet allocation; Social media; Law enforcement; Terrorism; Computer science; Data science; Topic model; Set (abstract data type); Event (particle physics); Computational sociology; Perception; Digital forensics; Computer security; Internet privacy; World Wide Web; Artificial intelligence; Political science; Law; Psychology","authors":[{"name":"Maxime Bérubé","is_ca":true},{"name":"Thuc-Uyên Tang","is_ca":true},{"name":"Francis Fortin","is_ca":true},{"name":"Sefa Ozalp","is_ca":false},{"name":"Matthew Williams","is_ca":false},{"name":"Pete Burnap","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1042927051340976,"gpt":0.4482298001236109,"spread":0.3439370949895133,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003593775,0.0005607992,0.0002524446,0.00544313,0.001907091,0.003316522,0.001000022,0.00196355,0.003196311],"category_scores_gemma":[0.01748198,0.0002347476,0.0002916697,0.002147771,0.001324696,0.002361777,0.002188016,0.00126329,0.0009152187],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009404701,"about_ca_system_score_gemma":0.001507515,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006261862,"about_ca_topic_score_gemma":0.009369467,"domain_scores_codex":[0.9978282,0.001045774,0.0001321583,0.0001889097,0.0005904489,0.0002144751],"domain_scores_gemma":[0.9898168,0.006046783,0.001179033,0.0006368366,0.001946395,0.0003741335],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001298481,0.001102079,0.1797677,0.00145194,0.000254433,0.02729017,0.06298859,0.02156707,0.02862051,0.06156319,0.03716255,0.5769332],"study_design_scores_gemma":[0.0001154128,0.0006719412,0.2125174,0.002523313,0.0003034956,0.01016619,0.1634397,0.3857413,0.0342919,0.06669198,0.1231557,0.0003815748],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8863124,0.00118201,0.04648968,0.009278778,0.0006905796,0.0007172949,0.001275283,0.0004462838,0.05360776],"genre_scores_gemma":[0.9660861,0.000694936,0.02698894,0.0002548484,0.0002290537,0.000113106,0.0003213595,0.00004468463,0.00526694],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006261862,"threshold_uncertainty_score":0.01900589,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4393100941","doi":"10.1007/978-3-031-54752-2","title":"AI-Generated Popular Culture","year":2024,"lang":"en","type":"book","venue":"","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":22,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Popular culture; Computer science; Sociology; Media studies","authors":[{"name":"Marcel Danesi","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04177498592570473,"gpt":0.3866558812883201,"spread":0.3448808953626153,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006334646,0.0004052783,0.0002556778,0.001930584,0.004069481,0.01008495,0.0008152312,0.001257471,0.06567901],"category_scores_gemma":[0.001954756,0.0001995015,0.0002741262,0.002596532,0.00403849,0.005719602,0.003108188,0.002398465,0.0106461],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00302679,"about_ca_system_score_gemma":0.001405268,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003752873,"about_ca_topic_score_gemma":0.007671706,"domain_scores_codex":[0.9993525,0.0001803467,0.00001530036,0.00007857737,0.000290932,0.00008234257],"domain_scores_gemma":[0.9993761,0.0002416045,0.00003520943,0.00008330341,0.0001536387,0.0001100243],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001240337,0.00002443787,0.0002112153,0.00005982703,0.000004889569,0.00008507184,0.004814794,0.0001109238,0.0001717211,0.8471773,0.1044747,0.04285285],"study_design_scores_gemma":[0.000004583025,0.000006283304,0.0003862586,0.00008757474,0.000003081928,0.00008891916,0.002786159,0.0002559435,0.0002315304,0.08398104,0.9121624,0.000006276153],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.003030113,0.001741706,0.000942286,0.003706944,0.0006052906,0.00001091452,0.0001057021,0.00006033581,0.9897967],"genre_scores_gemma":[0.1388344,0.005068135,0.002153643,0.002401788,0.0007908833,0.00006907732,0.0004647021,0.000372357,0.8498449],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.06567901,"threshold_uncertainty_score":0.2197182,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3194665160","doi":"10.18637/jss.v099.i02","title":"The <i>R</i> Package <b>sentometrics</b> to Compute, Aggregate, and Predict with Textual Sentiment","year":2021,"lang":"en","type":"article","venue":"Journal of Statistical Software","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":22,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Institut de Valorisation des Données; Innoviris; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"R package; Computer science; Aggregate (composite); Indexation; Sentiment analysis; Workflow; Index (typography); Volatility (finance); Series (stratigraphy); Information retrieval; Econometrics; Natural language processing; Mathematics; World Wide Web; Database; Programming language; Economics","authors":[{"name":"David Ardia","is_ca":false},{"name":"Keven Bluteau","is_ca":false},{"name":"Samuel Borms","is_ca":false},{"name":"Kris Boudt","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01761164827682454,"gpt":0.3292169960357477,"spread":0.3116053477589231,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00565223,0.003960526,0.002349967,0.004585919,0.001078929,0.004558153,0.00230381,0.001047344,0.1169578],"category_scores_gemma":[0.0400102,0.002216869,0.003313256,0.003604017,0.001142177,0.002673466,0.003600307,0.003539061,0.1314002],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009809092,"about_ca_system_score_gemma":0.003287036,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003885551,"about_ca_topic_score_gemma":0.005639477,"domain_scores_codex":[0.9967024,0.001075758,0.0003167398,0.0006857096,0.0009864455,0.0002329718],"domain_scores_gemma":[0.9871176,0.006993982,0.001299975,0.002451288,0.00176582,0.0003713866],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004305445,0.0000930461,0.007651246,0.002725386,0.001238498,0.0004298061,0.0004682694,0.005777791,0.00760073,0.0180459,0.8355663,0.1199725],"study_design_scores_gemma":[0.0004381232,0.0001891092,0.01047628,0.0006309347,0.0004817015,0.0008715601,0.0001871886,0.0475059,0.02273466,0.06493273,0.8510824,0.0004694561],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"software","genre_gemma":"software","genre_scores_codex":[0.003361946,0.0008165541,0.3883452,0.001069779,0.000885379,0.0004533298,0.11499,0.4799657,0.01011218],"genre_scores_gemma":[0.0347128,0.001234905,0.5230596,0.001906385,0.0005725621,0.004366043,0.1073245,0.3095235,0.01729964],"genre_candidate":"software","genre_consensus":"software","teacher_disagreement_score":0.1169578,"threshold_uncertainty_score":0.3912627,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4224311899","doi":"10.1111/cars.12378","title":"A new method for computational cultural cartography: From neural word embeddings to transformers and Bayesian mixture models","year":2022,"lang":"en","type":"article","venue":"Canadian Review of Sociology/Revue canadienne de sociologie","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":21,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Artificial intelligence; Computer science; Latent semantic analysis; Word (group theory); Natural language processing; Linguistics","authors":[{"name":"John McLevey","is_ca":true},{"name":"Tyler Crick","is_ca":true},{"name":"Pierson Browne","is_ca":true},{"name":"Darrin Durant","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05651616741544198,"gpt":0.3614541987538037,"spread":0.3049380313383618,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004159798,0.001231847,0.0009976198,0.003599072,0.0009475906,0.003413902,0.002248816,0.001284656,0.004602183],"category_scores_gemma":[0.02682796,0.0009304911,0.002199941,0.004199039,0.00310881,0.008011456,0.004458851,0.004096296,0.001300342],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001542588,"about_ca_system_score_gemma":0.001907649,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006425965,"about_ca_topic_score_gemma":0.006794915,"domain_scores_codex":[0.9971563,0.001739412,0.0001555826,0.0004659395,0.0003918096,0.00009095328],"domain_scores_gemma":[0.9920548,0.005787101,0.0004374079,0.0009944165,0.000561115,0.0001651249],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00009896181,0.00006164212,0.00227772,0.0002252969,0.0001951593,0.00009805348,0.001396574,0.08736521,0.001016672,0.6462542,0.004671749,0.2563388],"study_design_scores_gemma":[0.0000178482,0.00002001473,0.0002829843,0.00005208864,0.0000299847,0.00006498171,0.0001410447,0.4050152,0.0005543212,0.5875134,0.006277348,0.00003067014],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001485848,0.0001047535,0.9972991,0.0002227788,0.00002994566,0.00002246864,0.00008745935,0.0001913971,0.0005561823],"genre_scores_gemma":[0.1028384,0.0005240971,0.8925959,0.0002556872,0.0001408816,0.0004146356,0.0005861632,0.0003326497,0.002311487],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006425965,"threshold_uncertainty_score":0.02199936,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3121876164","doi":"10.1007/s11606-020-06579-3","title":"The Gender of COVID-19 Experts in Newspaper Articles: a Descriptive Cross-Sectional Study","year":2021,"lang":"en","type":"article","venue":"Journal of General Internal Medicine","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":20,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta; University of Calgary; University of British Columbia","funders":"Cumming School of Medicine, University of Calgary","keywords":"Newspaper; Medicine; Pandemic; Cross-sectional study; Confidence interval; Coronavirus disease 2019 (COVID-19); Family medicine; Media studies; Sociology; Pathology; Disease; Internal medicine","authors":[{"name":"Sarah Fletcher","is_ca":true},{"name":"Moss Bruton Joe","is_ca":true},{"name":"Santanna Hernandez","is_ca":true},{"name":"Inka Toman","is_ca":true},{"name":"Tyrone G. Harrison","is_ca":true},{"name":"Shannon M. Ruzycki","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1685033957827846,"gpt":0.4774958725633728,"spread":0.3089924767805882,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.001609184,0.0002580404,0.0004225157,0.002381588,0.0009854542,0.001908218,0.000445121,0.001033149,0.006695838],"category_scores_gemma":[0.009806242,0.0004948876,0.0004360093,0.001948717,0.000497736,0.001643781,0.0009106741,0.0009604293,0.001478615],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004473269,"about_ca_system_score_gemma":0.0004324099,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002636098,"about_ca_topic_score_gemma":0.004988891,"domain_scores_codex":[0.9982339,0.0004349872,0.0002855693,0.0002900216,0.000462994,0.0002924184],"domain_scores_gemma":[0.9892803,0.003409711,0.004433807,0.0004090024,0.001365701,0.001101589],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001308593,0.0002933482,0.9940898,0.00002937985,0.00003821819,0.0001212683,0.003278632,0.000006486823,0.0003375902,0.00003900626,0.000286646,0.001348609],"study_design_scores_gemma":[0.000004299531,0.0001192195,0.9904535,0.00001844032,0.00001554581,0.0002009421,0.00866833,0.00003454116,0.00005757847,0.00002441774,0.0003961197,0.000007085985],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9981847,0.0001189033,0.00004235206,0.00008175735,0.00001556189,0.00002192072,0.0005341079,0.000001787597,0.0009988821],"genre_scores_gemma":[0.9976735,0.0001258732,0.00008248646,0.0001355073,0.00003854992,0.00004855084,0.0005525011,0.0000068578,0.001336142],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9976184,"threshold_uncertainty_score":0.02239978,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2991039843","doi":"10.1075/ahs.10","title":"Keeping in Touch","year":2019,"lang":"en","type":"book","venue":"Advances in historical sociolinguistics","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":20,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Computer science","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.03113628056745455,"gpt":0.3771019742061559,"spread":0.3459656936387014,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001092959,0.0007217395,0.0007193304,0.001678629,0.003441613,0.01123692,0.001443089,0.002553899,0.1205038],"category_scores_gemma":[0.004575779,0.0003426185,0.0005998599,0.002014337,0.003725402,0.0103973,0.004770935,0.005612819,0.1072206],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001769744,"about_ca_system_score_gemma":0.002484652,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001916271,"about_ca_topic_score_gemma":0.0063242,"domain_scores_codex":[0.9984688,0.0002923603,0.00007205293,0.0002824154,0.0007691341,0.0001152707],"domain_scores_gemma":[0.9979248,0.0005979165,0.00008366584,0.0003772159,0.0006734774,0.0003429789],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002537749,0.00003111261,0.0002722201,0.0002924997,0.000009013448,0.0001105367,0.002515614,0.00008411466,0.0006586023,0.08806848,0.7127632,0.1951692],"study_design_scores_gemma":[6.03458e-7,0.000003381029,0.00006454592,0.00007335527,0.000001275287,0.00007308523,0.0002226434,0.000009024155,0.00005051712,0.004271546,0.9952276,0.000002337309],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.001660618,0.03768571,0.007657393,0.0571208,0.02603597,0.00005734398,0.0003812252,0.0008282183,0.8685728],"genre_scores_gemma":[0.006939954,0.0184508,0.004455876,0.01565908,0.00795495,0.00003752159,0.0003245558,0.0007498796,0.9454274],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.1205038,"threshold_uncertainty_score":0.4031253,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2889291357","doi":"10.1017/pan.2018.30","title":"Ideological Scaling of Social Media Users: A Dynamic Lexicon Approach","year":2018,"lang":"en","type":"article","venue":"Political Analysis","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":20,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto; Université Laval","funders":"University of Cambridge; Ministère de l'Économie, de la Science et de l'Innovation - Québec; Compute Canada; Université Laval","keywords":"Ideology; Social media; Voting behavior; Lexicon; Politics; Voting; Dimension (graph theory); Rhetoric; Sociology; Computer science; Political science; Linguistics; Artificial intelligence; Law; Mathematics; World Wide Web","authors":[{"name":"Mickael Temporão","is_ca":true},{"name":"Corentin Vande Kerckhove","is_ca":false},{"name":"Clifton van der Linden","is_ca":true},{"name":"Yannick Dufresne","is_ca":true},{"name":"Julien M. Hendrickx","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06638905123520183,"gpt":0.4087408830263073,"spread":0.3423518317911055,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003866584,0.0005807362,0.0008238272,0.008680145,0.001385953,0.004835264,0.001129242,0.0008682817,0.003152896],"category_scores_gemma":[0.03665362,0.0005149004,0.0009683371,0.006090068,0.001917576,0.005175007,0.003013638,0.001097053,0.0009227694],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00130956,"about_ca_system_score_gemma":0.00085068,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002407346,"about_ca_topic_score_gemma":0.002708683,"domain_scores_codex":[0.9956117,0.002326533,0.0003033253,0.0009441947,0.0006386566,0.0001755778],"domain_scores_gemma":[0.9815302,0.01392238,0.001593763,0.001632288,0.001040803,0.0002805899],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005615852,0.000858164,0.1361683,0.0005909087,0.0005833551,0.0006411146,0.02042319,0.02346293,0.01008712,0.2924798,0.006342283,0.5078012],"study_design_scores_gemma":[0.00008528243,0.0001480172,0.07080494,0.0001591556,0.0001413834,0.0004275903,0.00760382,0.5573376,0.002633529,0.3472471,0.01326479,0.000146707],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3650187,0.0004778806,0.6043807,0.001283666,0.0001089236,0.0006709094,0.003079311,0.0006060119,0.0243739],"genre_scores_gemma":[0.8863655,0.0001460762,0.1094967,0.00009466286,0.0001000836,0.0006667295,0.001689114,0.00009868506,0.001342519],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008680145,"threshold_uncertainty_score":0.02044868,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3111126926","doi":"10.1007/s11229-020-02915-6","title":"Eight journals over eight decades: a computational topic-modeling approach to contemporary philosophy of science","year":2020,"lang":"en","type":"article","venue":"Synthese","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":20,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Université du Québec à Montréal","funders":"Canada Research Chairs; Canada Foundation for Innovation","keywords":"Philosophy of science; Philosophy of language; Philosophy of computer science; Epistemology; History and philosophy of science; Empiricism; Philosophy of biology; Field (mathematics); Object (grammar); Western philosophy; Metaphysics; Social science; Sociology; Philosophy; Mathematics; Linguistics","authors":[{"name":"Christophe Malaterre","is_ca":true},{"name":"Francis Lareau","is_ca":true},{"name":"Davide Pulizzotto","is_ca":true},{"name":"Jonathan St-Onge","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1796726069853583,"gpt":0.4014957260038886,"spread":0.2218231190185303,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics","sts"],"consensus_categories":[],"category_scores_codex":[0.012802,0.0004500322,0.0006472298,0.0232338,0.0029038,0.01082175,0.001263386,0.002107086,0.008134044],"category_scores_gemma":[0.04315998,0.0004476298,0.000615951,0.03159397,0.003392135,0.01299128,0.002441322,0.002074912,0.0007943144],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003661669,"about_ca_system_score_gemma":0.00654312,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002430002,"about_ca_topic_score_gemma":0.0036261,"domain_scores_codex":[0.9945307,0.00296415,0.0004583859,0.0007490962,0.001104527,0.0001932327],"domain_scores_gemma":[0.9242824,0.06190292,0.005502461,0.001860336,0.004646746,0.001805095],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.0002309477,0.0001375796,0.01956108,0.002534507,0.0003293077,0.0003446424,0.04401905,0.002819597,0.001010363,0.6401717,0.02770907,0.2611322],"study_design_scores_gemma":[0.0001019653,0.000126502,0.02530216,0.003839946,0.0004455765,0.0003564275,0.03023617,0.01278387,0.001199435,0.5500785,0.3754385,0.00009099192],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.3111036,0.1369717,0.1654176,0.1550625,0.005089028,0.0004998985,0.00752263,0.001246877,0.2170863],"genre_scores_gemma":[0.8881558,0.03162785,0.05990591,0.003500312,0.002948717,0.000493651,0.002556234,0.0003194879,0.01049195],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9970962,"threshold_uncertainty_score":0.06770426,"prediction_status":"machine_predicted_unvalidated"},"labels":[{"model":"gemma","categories":["bibliometrics","sts"],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":["bibliometrics"],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W3165231223","doi":"10.1037/amp0000903","title":"Expert predictions of societal change: Insights from the world after COVID project.","year":2021,"lang":"en","type":"article","venue":"American Psychologist","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":19,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Ontario Ministry of Research and Innovation; Social Sciences and Humanities Research Council of Canada; Templeton World Charity Foundation","keywords":"PsycINFO; Pandemic; Coronavirus disease 2019 (COVID-19); Psychology; Value (mathematics); Politics; Scientific consensus; Set (abstract data type); Social psychology; Public relations; Sociology; Political science; MEDLINE; Law; Medicine; Computer science","authors":[{"name":"Igor Grossmann","is_ca":true},{"name":"Oliver Twardus","is_ca":true},{"name":"Michael E. W. Varnum","is_ca":false},{"name":"Eranda Jayawickreme","is_ca":false},{"name":"John McLevey","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1072679459642904,"gpt":0.4593135807908237,"spread":0.3520456348265334,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01948185,0.0004437801,0.0002558964,0.001867999,0.006249996,0.005986377,0.001931,0.002902316,0.002885855],"category_scores_gemma":[0.05934578,0.0003327153,0.0002620459,0.002132761,0.004915408,0.007850576,0.005643624,0.004200669,0.0004738758],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005111522,"about_ca_system_score_gemma":0.003408341,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01985,"about_ca_topic_score_gemma":0.02781427,"domain_scores_codex":[0.9874489,0.01006343,0.0002693554,0.0006160952,0.0009088684,0.0006932697],"domain_scores_gemma":[0.9102501,0.08115672,0.003187808,0.001214059,0.002699247,0.001492165],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.00005513892,0.00005578846,0.01166702,0.0001486115,0.00001313544,0.0006681925,0.9505899,0.0004372675,0.0002106704,0.00647443,0.008753377,0.02092633],"study_design_scores_gemma":[0.000005782183,0.00002644469,0.008994631,0.0002696358,0.000005437378,0.0001214821,0.9473814,0.001586713,0.0002163591,0.006849004,0.03451053,0.00003262535],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8954669,0.001625445,0.009955938,0.03678089,0.000175464,0.0002729875,0.001686273,0.00007427404,0.05396188],"genre_scores_gemma":[0.9905084,0.0007662077,0.002982383,0.001903015,0.00004478839,0.0002206451,0.0007619464,0.00004699584,0.002765594],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01985,"threshold_uncertainty_score":0.1030311,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}