{"meta":{"query_hash":"1f0d90f13f76","filters":{"venue":"Empirical Software Engineering"},"cohort_total":371,"direct_labels_cover":1,"predictions_cover":371,"exported":371,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/1f0d90f13f76","api":"https://metacan.xera.ac/api/v1/cohort?venue=Empirical+Software+Engineering"},"results":[{"id":"W1024906986","doi":"10.1007/s10664-015-9387-3","title":"A contextual approach towards more accurate duplicate bug report detection and ranking","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":90,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Data deduplication; Software bug; Android (operating system); Software; Ranking (information retrieval); Context (archaeology); Data science; Software engineering; Data mining; World Wide Web; Information retrieval; Database","score_opus":0.05477320630831465,"score_gpt":0.3061968965378645,"score_spread":0.25142369022954986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1024906986","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046207752,0.0019917546,0.93771666,0.0008581258,0.00032150486,0.00038491536,0.0018881423,0.0071989605,0.0034321365],"genre_scores_gemma":[0.29234838,0.0005453317,0.70032775,0.00045024185,0.00053776824,0.00030801856,0.002719136,0.00053159177,0.0022317164],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98705286,0.003792562,0.0012008535,0.0029276395,0.004370277,0.0006557617],"domain_scores_gemma":[0.96314037,0.011398085,0.0038102854,0.009844602,0.010938525,0.00086806295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055445614,0.0016466171,0.0030921355,0.009286934,0.0018235494,0.0041617835,0.0028754016,0.002600113,0.0044913343],"category_scores_gemma":[0.03724048,0.00093015924,0.0014202576,0.007033832,0.00093471457,0.0041001844,0.004323479,0.0022933108,0.002688098],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011199029,0.0012526659,0.03931684,0.0013809397,0.00050587236,0.00068157335,0.0010973972,0.02977865,0.067421034,0.019268462,0.025200605,0.81297606],"study_design_scores_gemma":[0.00026428766,0.0013374211,0.032803357,0.00043058503,0.00093393447,0.0018495712,0.0012586724,0.7933056,0.06508187,0.05318017,0.049160376,0.00039421127],"about_ca_topic_score_codex":0.005432454,"about_ca_topic_score_gemma":0.016894316,"teacher_disagreement_score":0.009286934,"about_ca_system_score_codex":0.00084165606,"about_ca_system_score_gemma":0.0041713812,"threshold_uncertainty_score":0.029322803},"labels":[],"label_agreement":null},{"id":"W1174118978","doi":"10.1007/s10664-015-9388-2","title":"Fresh apps: an empirical study of frequently-updated mobile apps in the Google play store","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Digital Marketing and Social Media","field":"Social Sciences","cited_by":201,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Queen's University","funders":"","keywords":"Mobile apps; Computer science; World Wide Web; App store; Internet privacy","score_opus":0.049901752931140124,"score_gpt":0.34997055647660247,"score_spread":0.30006880354546234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1174118978","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9983158,0.000059349055,0.0000450999,0.0001047198,0.0000044396565,0.00003702151,0.00009696736,0.0000049680934,0.0013316899],"genre_scores_gemma":[0.9968284,0.00016938549,0.00019589972,0.0001512777,0.00001699631,0.000045814468,0.00030428678,0.000021870923,0.002266113],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9984762,0.0003452973,0.00011772274,0.0001728789,0.00062487536,0.00026300494],"domain_scores_gemma":[0.96809196,0.01887683,0.0057702586,0.0012269085,0.0036380196,0.002396017],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021687122,0.0004925822,0.00047238835,0.003439503,0.0027569223,0.0053564883,0.001325884,0.0016721085,0.0062054046],"category_scores_gemma":[0.028063757,0.0007929093,0.00042650336,0.0029371448,0.0023826337,0.007089843,0.0022792404,0.0027340204,0.0013852774],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004477452,0.003887026,0.84341854,0.0001751132,0.00009137182,0.00096708216,0.12510522,0.00008408011,0.00081809383,0.0016735501,0.00238379,0.02094831],"study_design_scores_gemma":[0.000043682732,0.0005065112,0.7996266,0.00014910466,0.00011523595,0.00044234275,0.19243543,0.00082814426,0.0003933659,0.00039747573,0.00498613,0.00007593937],"about_ca_topic_score_codex":0.036375947,"about_ca_topic_score_gemma":0.07077061,"teacher_disagreement_score":0.036375947,"about_ca_system_score_codex":0.0015054939,"about_ca_system_score_gemma":0.0023322394,"threshold_uncertainty_score":0.07232839},"labels":[],"label_agreement":null},{"id":"W1260237309","doi":"10.1007/s10664-015-9364-x","title":"Coevolution of variability models and related software artifacts","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Software evolution; Coevolution; Variation (astronomy); Artifact (error); Computer science; Software; Feature (linguistics); Representation (politics); Personalization; Software system; Kernel (algebra); Data science; Machine learning; Artificial intelligence; Data mining; Software construction","score_opus":0.07639605922882918,"score_gpt":0.30006684688225754,"score_spread":0.22367078765342835,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1260237309","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6075577,0.00023558362,0.38357165,0.0004354133,0.000022949973,0.000059160742,0.00008854389,0.00023046095,0.007798609],"genre_scores_gemma":[0.968112,0.00011958052,0.030027669,0.000030982694,0.000007848016,0.000044893444,0.00009223913,0.00006782302,0.0014970768],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9966719,0.0019367124,0.000116854906,0.00055662746,0.00056323793,0.00015456743],"domain_scores_gemma":[0.97033477,0.021845287,0.0020971682,0.003968419,0.0012654891,0.0004889345],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048821312,0.00048912247,0.0005511622,0.0017688172,0.0005169363,0.0029936284,0.0011925816,0.001254211,0.0020224352],"category_scores_gemma":[0.04869371,0.0006388811,0.00091994886,0.0010450134,0.0013828474,0.0036653625,0.0018552417,0.001614355,0.0002741981],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019900514,0.00029567516,0.06725051,0.00014900384,0.00042400326,0.0007215164,0.002025632,0.4159164,0.008680425,0.42262736,0.0009167822,0.0807937],"study_design_scores_gemma":[0.000018349538,0.00009404873,0.010113771,0.000029357594,0.00008417009,0.00033216344,0.0002505666,0.8269166,0.0012819801,0.15956281,0.0012805046,0.00003580693],"about_ca_topic_score_codex":0.0017634378,"about_ca_topic_score_gemma":0.0017145755,"teacher_disagreement_score":0.0048821312,"about_ca_system_score_codex":0.0010778846,"about_ca_system_score_gemma":0.00072280265,"threshold_uncertainty_score":0.02581948},"labels":[],"label_agreement":null},{"id":"W1491417663","doi":"10.1023/a:1009859121816","title":"Ethics and Empirical Studies of Software Engineering","year":2000,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Ethics in Business and Education","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science","score_opus":0.311473937096708,"score_gpt":0.4635035632400121,"score_spread":0.15202962614330406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1491417663","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38774586,0.09212834,0.049316943,0.1602601,0.0009817643,0.00021680321,0.00019221379,0.00005414529,0.30910382],"genre_scores_gemma":[0.97927606,0.011471623,0.003244367,0.0027066164,0.00040610938,0.00012961795,0.000050605806,0.00001794851,0.0026970014],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97899604,0.017729606,0.00051598245,0.000518392,0.0018021239,0.0004378284],"domain_scores_gemma":[0.70156974,0.27949148,0.0070649646,0.0036947387,0.0063455626,0.0018335768],"candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.0144841485,0.00043904033,0.00043347196,0.0044623115,0.0027060953,0.004448918,0.00083066535,0.002653886,0.005209583],"category_scores_gemma":[0.1272418,0.00036013365,0.00017868193,0.003996321,0.022582538,0.008192836,0.0020979603,0.0033564172,0.00026639836],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030858475,0.00028908666,0.008271894,0.00038266619,0.000019664416,0.00009248326,0.0066630635,0.00072830415,0.00011867929,0.9588423,0.0028351187,0.021725953],"study_design_scores_gemma":[0.000030918218,0.000064480235,0.011461447,0.0013187322,0.000014396818,0.00019841999,0.017083354,0.0014209251,0.00024303784,0.9344148,0.03373357,0.000015944717],"about_ca_topic_score_codex":0.0028459488,"about_ca_topic_score_gemma":0.0030033821,"teacher_disagreement_score":0.9972939,"about_ca_system_score_codex":0.003111927,"about_ca_system_score_gemma":0.004270661,"threshold_uncertainty_score":0.07660043},"labels":[],"label_agreement":null},{"id":"W1493296086","doi":"10.1007/s10664-014-9308-x","title":"Understanding the impact of rapid releases on software quality","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":59,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Polytechnique Montréal","funders":"","keywords":"Computer science; Software release life cycle; Software engineering; Crash; Software quality; Software; Quality (philosophy); Software bug; Software quality analyst; Software quality assurance; Software development; Operating system","score_opus":0.11284751522180525,"score_gpt":0.3485465269864667,"score_spread":0.23569901176466146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1493296086","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9870481,0.0006787787,0.0019905353,0.0014675538,0.00001750796,0.000012565096,0.00010719434,0.000041481962,0.008636277],"genre_scores_gemma":[0.9986981,0.00021926571,0.00034957053,0.00006432055,0.000019238118,0.0000036127674,0.000048109217,0.000015984699,0.00058164774],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9958625,0.0018513605,0.00016376864,0.00035758744,0.0012474306,0.00051732967],"domain_scores_gemma":[0.8305765,0.13563432,0.020508679,0.0040833363,0.006711033,0.002486182],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006575145,0.00033545005,0.00022169376,0.0012466682,0.0003639849,0.002650788,0.00063172734,0.0007225048,0.0053715413],"category_scores_gemma":[0.082188606,0.0003142977,0.00034340506,0.0010755843,0.0009780439,0.0045724185,0.0010129744,0.0017584773,0.00046099882],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012849903,0.0020655645,0.73646396,0.00060333934,0.0004948434,0.0006678892,0.0035674218,0.030894596,0.0095191,0.030389568,0.003547351,0.18050137],"study_design_scores_gemma":[0.00008966751,0.0009472173,0.942774,0.000122031444,0.00025155288,0.00013419327,0.003019379,0.028830081,0.0027144544,0.01775897,0.0033086205,0.000049856204],"about_ca_topic_score_codex":0.007476118,"about_ca_topic_score_gemma":0.009481686,"teacher_disagreement_score":0.007476118,"about_ca_system_score_codex":0.0014474343,"about_ca_system_score_gemma":0.0013111045,"threshold_uncertainty_score":0.03477317},"labels":[],"label_agreement":null},{"id":"W1497968089","doi":"10.1023/a:1024424811345","title":"Fault Prediction Modeling for Software Quality Estimation: Comparing Commonly Used Techniques","year":2003,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Reliability and Analysis Research","field":"Computer Science","cited_by":128,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"McGill University","keywords":"Computer science; Software quality; Software; Artificial neural network; Approximation error; Reliability engineering; Data mining; Artificial intelligence; Machine learning; Algorithm; Software development; Engineering","score_opus":0.07721747522635733,"score_gpt":0.35017777725256943,"score_spread":0.27296030202621213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1497968089","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21866773,0.012655563,0.76220876,0.00094459165,0.00013717714,0.00015337873,0.0006193991,0.0021656526,0.0024477988],"genre_scores_gemma":[0.85369027,0.006941518,0.13744397,0.00013677037,0.00015481818,0.00016804489,0.0007252484,0.0001933283,0.00054607174],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9945322,0.0025507193,0.00041727975,0.0005544063,0.0017247454,0.00022048713],"domain_scores_gemma":[0.93857986,0.051068667,0.00302878,0.0027968928,0.004254357,0.0002713654],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009127711,0.0013881788,0.0019140325,0.0060726735,0.0005506693,0.0013600638,0.0026719423,0.0015183411,0.001093708],"category_scores_gemma":[0.041701686,0.00046261703,0.0016953916,0.004516779,0.00055098673,0.0034617668,0.0013523562,0.001327667,0.00035987352],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012868625,0.0005207744,0.025829248,0.00095655804,0.0011524558,0.000050501447,0.0003388249,0.3516938,0.0008224197,0.0052317553,0.0015558471,0.6105609],"study_design_scores_gemma":[0.00007871245,0.00047651134,0.008125781,0.00014738654,0.00038563812,0.000084320585,0.00016057125,0.97939634,0.000921906,0.009486163,0.0006949401,0.000041755975],"about_ca_topic_score_codex":0.0109106405,"about_ca_topic_score_gemma":0.0076578218,"teacher_disagreement_score":0.0109106405,"about_ca_system_score_codex":0.0011677047,"about_ca_system_score_gemma":0.001221229,"threshold_uncertainty_score":0.04827249},"labels":[],"label_agreement":null},{"id":"W1507760270","doi":"10.1023/a:1011966430523","title":"Getting to the Source of Ethical Issues","year":2001,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Social and Cultural Studies","field":"Social Sciences","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science","score_opus":0.04003005214578316,"score_gpt":0.3469797730443074,"score_spread":0.30694972089852424,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1507760270","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06255041,0.0058566593,0.091066584,0.76159465,0.0052390145,0.00036509396,0.00012343875,0.00015339494,0.07305068],"genre_scores_gemma":[0.83024335,0.0051336186,0.045307335,0.10103992,0.0030298275,0.0006724819,0.00015394345,0.0005631382,0.013856408],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.8137064,0.1351025,0.0057602385,0.005882193,0.033924617,0.0056241336],"domain_scores_gemma":[0.35453743,0.47192454,0.032695964,0.045857336,0.08485622,0.010128431],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1391605,0.0009825119,0.0010031115,0.0062803454,0.011924985,0.020238634,0.003192005,0.017815707,0.011074092],"category_scores_gemma":[0.56442314,0.0010340248,0.0012679591,0.0038519057,0.03347569,0.034436803,0.012365759,0.035104703,0.0018092393],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000751637,0.00028188055,0.013561742,0.00056251703,0.00016312384,0.0011429725,0.09845104,0.00054444635,0.0012817727,0.69998443,0.069401965,0.11454895],"study_design_scores_gemma":[0.00006896644,0.00006065315,0.004684986,0.0027387773,0.00006919122,0.0007824747,0.10669332,0.0018912491,0.0021442433,0.7463192,0.13445196,0.00009494361],"about_ca_topic_score_codex":0.0029381574,"about_ca_topic_score_gemma":0.0035386765,"teacher_disagreement_score":0.8608395,"about_ca_system_score_codex":0.008079918,"about_ca_system_score_gemma":0.017343746,"threshold_uncertainty_score":0.7359599},"labels":[],"label_agreement":null},{"id":"W1549553848","doi":"10.1023/a:1009815306478","title":"Replicated Case Studies for Investigating Quality Factors in Object-Oriented Designs","year":2001,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":172,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Carleton University","funders":"","keywords":"Computer science; Cohesion (chemistry); Software engineering; Quality (philosophy); Software quality; Set (abstract data type); Software; Data science; Data mining; Programming language; Software development","score_opus":0.1714365202462394,"score_gpt":0.4017966966694224,"score_spread":0.23036017642318302,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1549553848","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85220027,0.0010675165,0.13643506,0.0002688266,0.00007932439,0.0026695651,0.00020846067,0.0001720441,0.0068987603],"genre_scores_gemma":[0.8847176,0.00038568937,0.11228758,0.00007154132,0.000029008836,0.0015907215,0.00015007198,0.000030516461,0.0007373183],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9327002,0.05449029,0.002794588,0.0019036072,0.0075100306,0.00060136773],"domain_scores_gemma":[0.5909102,0.31444126,0.018235423,0.058225106,0.016257698,0.0019302604],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03770876,0.000669368,0.0007544509,0.0028995243,0.0016648599,0.0020272657,0.0031659368,0.0025314076,0.0033028196],"category_scores_gemma":[0.23253487,0.00077361707,0.0010804264,0.0024994335,0.0020305736,0.0036264067,0.002375607,0.0013830292,0.00032908586],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0069904695,0.024842946,0.34752685,0.0042994823,0.002162767,0.0066738348,0.049685966,0.03797113,0.043212358,0.09964473,0.0028753725,0.37411416],"study_design_scores_gemma":[0.008252302,0.06396508,0.2910159,0.002443034,0.0044408953,0.011118627,0.054018922,0.27219704,0.076450616,0.18501565,0.030254671,0.000827237],"about_ca_topic_score_codex":0.0024738442,"about_ca_topic_score_gemma":0.005591773,"teacher_disagreement_score":0.96229124,"about_ca_system_score_codex":0.0021167777,"about_ca_system_score_gemma":0.0021471914,"threshold_uncertainty_score":0.19942534},"labels":[],"label_agreement":null},{"id":"W1564555743","doi":"10.1023/a:1011998412776","title":"Why and How Research Ethics Matters to You. Yes, You!","year":2001,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Ethics in Clinical Research","field":"Medicine","cited_by":29,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Medicine","score_opus":0.5129399488320844,"score_gpt":0.562356893618366,"score_spread":0.04941694478628167,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1564555743","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010076732,0.012051146,0.007616064,0.96833324,0.004554429,0.00003072756,0.000034733388,0.0000625748,0.006309428],"genre_scores_gemma":[0.1258581,0.04951412,0.037813812,0.7437102,0.01925287,0.00044802635,0.00015561272,0.00061110064,0.02263612],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.85633415,0.10707625,0.006005397,0.004190117,0.023427648,0.00296645],"domain_scores_gemma":[0.6271248,0.24900703,0.017432097,0.020379458,0.06992986,0.016126744],"candidate_categories":["metaresearch","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.109163746,0.0008508782,0.0015733315,0.0026015986,0.00618319,0.016599614,0.0018073071,0.013475614,0.007921032],"category_scores_gemma":[0.358449,0.0008356837,0.0010408444,0.0024079655,0.046411663,0.023782264,0.0064087627,0.022943856,0.006897495],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001611967,0.00016451813,0.005434607,0.0018577628,0.00026823595,0.00036068785,0.014799157,0.00037803978,0.001016131,0.32345212,0.53359294,0.11851472],"study_design_scores_gemma":[0.00007570869,0.00012941321,0.002784846,0.004061676,0.00009301872,0.0010568746,0.014522967,0.00046181123,0.00091541716,0.46577856,0.50996417,0.0001555232],"about_ca_topic_score_codex":0.0040767496,"about_ca_topic_score_gemma":0.0051971083,"teacher_disagreement_score":0.9865244,"about_ca_system_score_codex":0.0055934247,"about_ca_system_score_gemma":0.015481407,"threshold_uncertainty_score":0.57732},"labels":[],"label_agreement":null},{"id":"W1576984811","doi":"10.1023/a:1025320418915","title":"An Externally Replicated Experiment for Evaluating the Learning Effectiveness of Using Simulations in Software Project Management Education","year":2003,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"European Commission; Technische Universität Kaiserslautern","keywords":"Computer science; COCOMO; Debriefing; Software project management; Empirical research; Project management; Software; Software development; Process (computing); Software engineering; Knowledge management; Engineering management; Engineering; Systems engineering; Software construction; Psychology","score_opus":0.07108431282324686,"score_gpt":0.42070935125680736,"score_spread":0.3496250384335605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1576984811","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98685294,0.000028791228,0.0077241347,0.0000708632,0.00016370654,0.0027957668,0.00015744075,0.000146323,0.0020599824],"genre_scores_gemma":[0.96532184,0.00005926689,0.02300745,0.00013238749,0.00008900061,0.008364758,0.00024759932,0.000063331194,0.0027145648],"study_design_codex":"nonrandomized_trial","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9845834,0.0095525,0.0015241839,0.002010467,0.001831183,0.00049822475],"domain_scores_gemma":[0.8609152,0.099177174,0.008730609,0.019781072,0.007433842,0.0039620693],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.015933555,0.001635989,0.0013257804,0.00077476434,0.001271505,0.0016395631,0.002849667,0.0031940935,0.004298294],"category_scores_gemma":[0.0868552,0.0009666419,0.0007350606,0.00052704645,0.0017592416,0.0017354682,0.0018425207,0.0027508459,0.00097351545],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.1841163,0.46683684,0.027218943,0.0013742907,0.00078179216,0.0003896489,0.006694447,0.026258517,0.17149977,0.0048180963,0.0015944771,0.108416796],"study_design_scores_gemma":[0.06618019,0.69301736,0.049552824,0.00026599452,0.0011623172,0.000236074,0.0013713664,0.042075746,0.13439745,0.0057447013,0.0055871923,0.0004087381],"about_ca_topic_score_codex":0.0007517543,"about_ca_topic_score_gemma":0.000722752,"teacher_disagreement_score":0.9840664,"about_ca_system_score_codex":0.0010010907,"about_ca_system_score_gemma":0.0023216235,"threshold_uncertainty_score":0.08426571},"labels":[],"label_agreement":null},{"id":"W1581639473","doi":"10.1007/s10664-012-9209-9","title":"Automated topic naming","year":2012,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of British Columbia; University of Alberta","funders":"","keywords":"Computer science; Latent Dirichlet allocation; Topic model; Commit; Categorization; Software; Context (archaeology); Information retrieval; Domain (mathematical analysis); Artificial intelligence; Natural language processing; Data science; Database; Programming language","score_opus":0.0256654668367926,"score_gpt":0.2953701151461752,"score_spread":0.2697046483093826,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1581639473","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02163064,0.0011782096,0.8201336,0.0010909581,0.0013558154,0.0008249888,0.015188715,0.111389354,0.02720772],"genre_scores_gemma":[0.16100122,0.0006583045,0.7594329,0.00033076367,0.0006044373,0.00074363506,0.039361235,0.009452349,0.028415095],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.993664,0.0017912894,0.0007036392,0.0017550412,0.0015183033,0.00056766754],"domain_scores_gemma":[0.98501855,0.0051109344,0.0006881762,0.0046881256,0.0037404753,0.0007536874],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003726395,0.0017770369,0.002168254,0.0102029415,0.0035137648,0.0067146043,0.0024372602,0.0017481969,0.05306211],"category_scores_gemma":[0.019340817,0.0011083531,0.0025482895,0.0059882747,0.0008304082,0.008740413,0.006092536,0.0022251224,0.036676116],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006562417,0.00018027984,0.0041445782,0.0007872213,0.0001544787,0.00026513234,0.0013302565,0.001746112,0.038478833,0.032340344,0.13964218,0.7802744],"study_design_scores_gemma":[0.00027359443,0.00020470943,0.00797001,0.00030973024,0.00037075125,0.0015230809,0.0032590395,0.19447966,0.091416515,0.13495828,0.56495285,0.0002819064],"about_ca_topic_score_codex":0.0032494674,"about_ca_topic_score_gemma":0.0050712693,"teacher_disagreement_score":0.05306211,"about_ca_system_score_codex":0.0017552067,"about_ca_system_score_gemma":0.0035841744,"threshold_uncertainty_score":0.17751044},"labels":[],"label_agreement":null},{"id":"W1589080523","doi":"10.1023/a:1023014713253","title":"An Investigation on the Occurrence of Service Requests in Commercial Software Applications","year":2003,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Reliability and Analysis Research","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Process (computing); Software quality; Service (business); Identification (biology); Software; Data mining; Reliability engineering; Software development; Engineering","score_opus":0.05197182144963392,"score_gpt":0.3105143093181651,"score_spread":0.2585424878685312,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1589080523","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9964799,0.00012653864,0.0012319143,0.00016841138,0.000009863399,0.00002893239,0.00011548392,0.000051631258,0.0017873473],"genre_scores_gemma":[0.9987274,0.00009132711,0.00058529864,0.00003394558,0.000018340266,0.000013772665,0.00012553633,0.00001128072,0.0003929603],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9959567,0.0013088088,0.00043015988,0.00028890063,0.0017277314,0.00028763412],"domain_scores_gemma":[0.9047314,0.070885494,0.012337127,0.0026554468,0.0077228807,0.0016676212],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026115743,0.00023238575,0.00023556758,0.003046792,0.0010762304,0.0010488983,0.0007395793,0.0010016522,0.0014480851],"category_scores_gemma":[0.04396782,0.00031913968,0.00030867103,0.0033659423,0.00069102563,0.0014191484,0.00058439,0.001104373,0.00042883182],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011120099,0.00061410165,0.9193781,0.0002474006,0.000068254114,0.0022062503,0.010201578,0.0013748824,0.009191916,0.0035007547,0.0013848052,0.050720077],"study_design_scores_gemma":[0.00003350703,0.0008777636,0.93889046,0.00007509796,0.00011899012,0.005101526,0.012939189,0.026829781,0.006949416,0.0024295838,0.0056880442,0.00006673061],"about_ca_topic_score_codex":0.0067050606,"about_ca_topic_score_gemma":0.006650906,"teacher_disagreement_score":0.0067050606,"about_ca_system_score_codex":0.0010249375,"about_ca_system_score_gemma":0.0013011012,"threshold_uncertainty_score":0.013811529},"labels":[],"label_agreement":null},{"id":"W1600717382","doi":"10.1007/s10664-012-9200-5","title":"Understanding Ajax applications by connecting client and server-side execution traces","year":2012,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Ajax; Computer science; World Wide Web; Web application; Client-side; Web 2.0; Web-based simulation; Field (mathematics); Web page; Web API","score_opus":0.06906239645443948,"score_gpt":0.29564332158658657,"score_spread":0.2265809251321471,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1600717382","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.52774197,0.00026890397,0.4459553,0.00042606535,0.000027847409,0.00048933097,0.0008011542,0.017403172,0.0068861833],"genre_scores_gemma":[0.75537395,0.00037758143,0.2381664,0.00008436743,0.000016783322,0.00017732986,0.0014063946,0.0013024167,0.0030948196],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99872357,0.00033905936,0.00012906826,0.00022343828,0.00051178125,0.000073032905],"domain_scores_gemma":[0.9836246,0.011693351,0.0012971299,0.0011299528,0.0020079059,0.00024707618],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021071695,0.0009374745,0.00031192513,0.002182763,0.00051316834,0.0021191684,0.00093985564,0.0006956467,0.0021296649],"category_scores_gemma":[0.02048454,0.00052726077,0.00024108126,0.000910001,0.0004741037,0.0051003313,0.0012086228,0.0011763738,0.00052880164],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011904283,0.0012526904,0.10450852,0.0012554155,0.00013890896,0.0017683078,0.048650954,0.03183965,0.13438234,0.014580723,0.0063667493,0.6540654],"study_design_scores_gemma":[0.000110487716,0.00086862606,0.12060619,0.00047303573,0.00019006382,0.0018883122,0.010508142,0.67108214,0.1166242,0.032600842,0.044753972,0.00029398527],"about_ca_topic_score_codex":0.0053046946,"about_ca_topic_score_gemma":0.0067311362,"teacher_disagreement_score":0.0053046946,"about_ca_system_score_codex":0.00046218437,"about_ca_system_score_gemma":0.0010349137,"threshold_uncertainty_score":0.011143923},"labels":[],"label_agreement":null},{"id":"W1774183706","doi":"10.1023/a:1011485205161","title":"Quantitative Measurements of the Influence of Participant Roles during Peer Review Meetings","year":2001,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Innovative Teaching and Learning Methods","field":"Psychology","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Division of Materials Research","keywords":"Artifact (error); Computer science; Protocol (science); Interpretation (philosophy); Content analysis; Focus (optics); Supervisor; Task (project management); Data science; Quality (philosophy); Knowledge management; Artificial intelligence; Engineering; Epistemology; Management","score_opus":0.1633991765249903,"score_gpt":0.4226693604507548,"score_spread":0.2592701839257645,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1774183706","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9899436,0.00035375755,0.0034730025,0.00013545719,0.00007007443,0.00021682607,0.00011312637,0.00006637095,0.005627839],"genre_scores_gemma":[0.9954621,0.00010249819,0.0021265573,0.000044395925,0.000076725075,0.00025246255,0.0001239884,0.00003137628,0.0017798358],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.97559696,0.013926219,0.0009485537,0.0019260868,0.006372924,0.0012293261],"domain_scores_gemma":[0.65079963,0.28631347,0.021344012,0.011172435,0.019093433,0.011276922],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.012436951,0.0005030377,0.00057550427,0.0014061341,0.0015845409,0.001540339,0.0009596075,0.0010533475,0.0032390838],"category_scores_gemma":[0.12934484,0.00035469868,0.00029396333,0.00069885683,0.0009849289,0.000689535,0.0013185499,0.0012970645,0.0007954418],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.009258581,0.005599111,0.21995415,0.0016655468,0.00047445882,0.0012287926,0.06729717,0.001103227,0.4704177,0.0024113709,0.0038219432,0.21676792],"study_design_scores_gemma":[0.00044873115,0.015827632,0.8100559,0.00020299027,0.00038143818,0.0014906503,0.020728095,0.003929705,0.1295289,0.002184094,0.014984791,0.0002370765],"about_ca_topic_score_codex":0.0005155644,"about_ca_topic_score_gemma":0.0009398893,"teacher_disagreement_score":0.9875631,"about_ca_system_score_codex":0.0004902888,"about_ca_system_score_gemma":0.0010501627,"threshold_uncertainty_score":0.065773726},"labels":[],"label_agreement":null},{"id":"W1838805003","doi":"10.1023/a:1011487332587","title":"Modelling the Likelihood of Software Process Improvement: An Exploratory Study","year":2001,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Capability Maturity Model; Computer science; Sample (material); Process (computing); Empirical research; Software; Process management; Data science; Engineering; Mathematics; Statistics","score_opus":0.03668895162732284,"score_gpt":0.2952584062826417,"score_spread":0.2585694546553189,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1838805003","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9691715,0.00015059828,0.028472122,0.00031756787,0.000004874617,0.00014341768,0.00018843522,0.00006961386,0.0014818433],"genre_scores_gemma":[0.99219257,0.000091354406,0.0069696074,0.000019185649,0.000007617621,0.00008732017,0.00019585565,0.000021717184,0.00041489798],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99195707,0.0063294573,0.00027284684,0.0005548014,0.0005304742,0.0003553474],"domain_scores_gemma":[0.36235157,0.6239454,0.006805261,0.003714147,0.002290761,0.0008928178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02029321,0.00085393054,0.00086406345,0.0014615762,0.0007746857,0.0028621953,0.002689255,0.0029585229,0.0045345686],"category_scores_gemma":[0.22296233,0.00082886114,0.0013274931,0.0017754339,0.0010860023,0.004448671,0.0014092382,0.00341611,0.00054186955],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0057912413,0.003632703,0.49722782,0.00063883397,0.00059523305,0.0011176986,0.007888073,0.3939619,0.001902491,0.027187834,0.0010170746,0.05903912],"study_design_scores_gemma":[0.00022915893,0.0016524511,0.037451126,0.00006171294,0.00022626438,0.00045892558,0.0013379672,0.94161344,0.0012433984,0.014914614,0.00073519425,0.00007584092],"about_ca_topic_score_codex":0.006553068,"about_ca_topic_score_gemma":0.0040925974,"teacher_disagreement_score":0.02029321,"about_ca_system_score_codex":0.0015531855,"about_ca_system_score_gemma":0.0015134607,"threshold_uncertainty_score":0.10732204},"labels":[],"label_agreement":null},{"id":"W1909497710","doi":"10.1007/s10664-015-9396-2","title":"Towards building a universal defect prediction model with rank transformed predictors","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":116,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"University of Victoria","keywords":"Computer science; Software; Rank (graph theory); Workflow; Predictive modelling; Context (archaeology); Eclipse; Data mining; Software development; Obstacle; Software bug; Software engineering; Machine learning; Database; Programming language; Mathematics","score_opus":0.02760184776548904,"score_gpt":0.2626829102734268,"score_spread":0.23508106250793773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1909497710","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031494536,0.0002372748,0.9655889,0.00030163702,0.000042664495,0.000047555983,0.0002721072,0.001329428,0.00068586366],"genre_scores_gemma":[0.58978766,0.00060711795,0.40218198,0.00038332978,0.00020293785,0.00025798884,0.0017893796,0.00028574548,0.00450384],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99805593,0.000593817,0.00015052508,0.00058825343,0.00039446022,0.0002171509],"domain_scores_gemma":[0.99412954,0.002699067,0.0005349945,0.0010987791,0.001313796,0.00022391217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046249214,0.0011733214,0.0021668072,0.0018499715,0.00059355254,0.001742359,0.0023338355,0.0015373337,0.001880942],"category_scores_gemma":[0.01195783,0.000783088,0.0013522485,0.0018841971,0.00098097,0.0031924692,0.0027558056,0.0023464027,0.001605855],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029182996,0.0005802307,0.023281325,0.0002691939,0.0004952291,0.00038424198,0.00035554304,0.5150514,0.008271501,0.039042983,0.006487436,0.40548903],"study_design_scores_gemma":[0.000008381884,0.000050160972,0.0008098582,0.00002082588,0.000040794504,0.00004603464,0.000020708576,0.98357683,0.0006385483,0.014135969,0.00063833763,0.000013570464],"about_ca_topic_score_codex":0.005129556,"about_ca_topic_score_gemma":0.0060565057,"teacher_disagreement_score":0.005129556,"about_ca_system_score_codex":0.000569749,"about_ca_system_score_gemma":0.002172701,"threshold_uncertainty_score":0.024459183},"labels":[],"label_agreement":null},{"id":"W1914969610","doi":"10.1007/s10664-015-9393-5","title":"An in-depth study of the promises and perils of mining GitHub","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":270,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Software; Point (geometry); Data science; World Wide Web; Set (abstract data type); Event (particle physics); Empirical research; The Internet; Quality (philosophy)","score_opus":0.05189227763191294,"score_gpt":0.31378800903066856,"score_spread":0.2618957313987556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1914969610","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85657966,0.015867228,0.06180893,0.027074505,0.00012936456,0.00029159663,0.00198889,0.00022928634,0.036030516],"genre_scores_gemma":[0.94508326,0.00616323,0.042049866,0.0011380754,0.00018276708,0.00007881095,0.0016292557,0.00011313947,0.0035616537],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.99071324,0.005087964,0.00045831472,0.00049070804,0.0028655084,0.00038424777],"domain_scores_gemma":[0.8384849,0.1324434,0.008217656,0.009278706,0.009914197,0.0016611865],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.014398425,0.0003185126,0.00037137818,0.004433096,0.0015059564,0.0045652315,0.0013795418,0.000858712,0.0022946477],"category_scores_gemma":[0.10823736,0.0003601127,0.00035523233,0.008001745,0.0023457387,0.010759787,0.0018934367,0.0019939437,0.0005213564],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047844206,0.00064161676,0.27630225,0.0023566908,0.00016052004,0.00032071557,0.016968518,0.0042813,0.0060032434,0.11944281,0.0134437485,0.5596002],"study_design_scores_gemma":[0.00006988324,0.0007331321,0.46164995,0.002946607,0.00020404876,0.0017454573,0.07258553,0.075227104,0.017628208,0.20010546,0.16689067,0.00021411003],"about_ca_topic_score_codex":0.0057446742,"about_ca_topic_score_gemma":0.018905,"teacher_disagreement_score":0.9856016,"about_ca_system_score_codex":0.0018978862,"about_ca_system_score_gemma":0.0038572466,"threshold_uncertainty_score":0.07614708},"labels":[],"label_agreement":null},{"id":"W1963574127","doi":"10.1007/s10664-014-9303-2","title":"Mining system logs to learn error predictors: a case study of a telemetry system","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Telemetry; Data mining; Computer science; Support vector machine; Robustness (evolution); Reliability (semiconductor); Word error rate; Software; Machine learning; Artificial intelligence","score_opus":0.0179966043784878,"score_gpt":0.2613391865108126,"score_spread":0.24334258213232482,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1963574127","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9793888,0.00013017202,0.018456565,0.00053758884,0.000014859909,0.00011889646,0.00033768322,0.00028286056,0.00073264836],"genre_scores_gemma":[0.97516733,0.000088683686,0.023604691,0.00004884336,0.000011545676,0.000033517856,0.00031762695,0.00003858853,0.0006892168],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9983961,0.00068194733,0.00014117904,0.0002032575,0.00046468148,0.000112902424],"domain_scores_gemma":[0.9655075,0.028698627,0.0013624597,0.0020422186,0.0019164699,0.00047273468],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029108343,0.0006866549,0.00040692286,0.001313949,0.0007111544,0.00089681405,0.0016720925,0.0016833704,0.00071456545],"category_scores_gemma":[0.018588226,0.00036531844,0.00038319515,0.0013011575,0.00061139185,0.0013583164,0.0006069876,0.0011973087,0.0002563452],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012243637,0.0076586576,0.42231914,0.0010893808,0.00036504393,0.015719684,0.0068496224,0.20728149,0.018658709,0.0027878617,0.0075026085,0.30854344],"study_design_scores_gemma":[0.00018953746,0.0015622025,0.09405541,0.000114750736,0.00016991778,0.003711057,0.004493571,0.8654204,0.021191608,0.0040790066,0.0049145846,0.00009792776],"about_ca_topic_score_codex":0.008615072,"about_ca_topic_score_gemma":0.014201089,"teacher_disagreement_score":0.008615072,"about_ca_system_score_codex":0.00061329413,"about_ca_system_score_gemma":0.0010400559,"threshold_uncertainty_score":0.017129838},"labels":[],"label_agreement":null},{"id":"W1969265968","doi":"10.1007/s10664-007-9054-4","title":"Analysis of attribute weighting heuristics for analogy-based software effort estimation method AQUA+","year":2007,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Weighting; Heuristics; Data mining; Set (abstract data type); Computer science; Selection (genetic algorithm); Estimation; Exploit; A-weighting; Machine learning; Engineering","score_opus":0.02959484286795332,"score_gpt":0.3421866429248011,"score_spread":0.3125918000568478,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1969265968","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4599824,0.00067774014,0.5317091,0.0002533361,0.0000602984,0.0003554786,0.0003126321,0.0017476715,0.0049013845],"genre_scores_gemma":[0.7402712,0.00008547048,0.25808296,0.000076376105,0.000013528166,0.00019341306,0.00038241557,0.000116999116,0.00077764667],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99480057,0.0033500695,0.00028362448,0.00050095486,0.0008875476,0.00017727335],"domain_scores_gemma":[0.95987296,0.033911604,0.0009830343,0.0021178694,0.002872254,0.0002422946],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007072041,0.00047825102,0.000793784,0.001614667,0.00061681104,0.0013645452,0.0014060664,0.0008045983,0.0030922838],"category_scores_gemma":[0.045704264,0.0003029659,0.0004887783,0.0016976771,0.00030541996,0.0021327285,0.0010838402,0.0009902479,0.000348295],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013759632,0.0009320373,0.029032068,0.00046869626,0.0002281688,0.00007973331,0.00095299736,0.05086263,0.0073087555,0.014916794,0.0038357826,0.89000636],"study_design_scores_gemma":[0.00012724378,0.00048681474,0.0104635,0.00005040032,0.0001631826,0.00013901581,0.00036223777,0.96813345,0.005691226,0.012261419,0.002078482,0.000043122218],"about_ca_topic_score_codex":0.002589422,"about_ca_topic_score_gemma":0.0030387957,"teacher_disagreement_score":0.007072041,"about_ca_system_score_codex":0.0007429414,"about_ca_system_score_gemma":0.0013832656,"threshold_uncertainty_score":0.03740102},"labels":[],"label_agreement":null},{"id":"W1975236650","doi":"10.1007/s10664-008-9101-9","title":"A subject-based empirical evaluation of SSUCD’s performance in reducing inconsistencies in use case models","year":2008,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":23,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Subject (documents); Empirical research; Statistics; Mathematics; World Wide Web","score_opus":0.4411743086357212,"score_gpt":0.4184957454988292,"score_spread":0.022678563136892038,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1975236650","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9365168,0.00068986445,0.05409384,0.0003718446,0.000083672254,0.000614785,0.0011515688,0.0019627516,0.0045148325],"genre_scores_gemma":[0.87922734,0.00013622594,0.11743681,0.000082608196,0.000024404037,0.00027331864,0.0021271699,0.00011300271,0.00057913753],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9749141,0.017411066,0.0027179222,0.0019784581,0.002580778,0.0003977379],"domain_scores_gemma":[0.6432277,0.29757187,0.008597016,0.024199877,0.024137087,0.0022665344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.047930658,0.0007116714,0.0009963879,0.0056472453,0.000952384,0.002449361,0.0022349355,0.0015741902,0.0018691862],"category_scores_gemma":[0.17834319,0.00039672456,0.00088824914,0.0035806955,0.0013171164,0.0034883686,0.0031395182,0.00088700047,0.00036741595],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008653301,0.0032131663,0.32389542,0.0012661574,0.00081205316,0.00020091962,0.0031322208,0.054431856,0.0055740746,0.0042795967,0.0041324417,0.59040874],"study_design_scores_gemma":[0.0015521751,0.0071251774,0.16814421,0.0002544254,0.0012041327,0.0005743662,0.0032773595,0.78192407,0.020090457,0.0051511712,0.010489985,0.00021243302],"about_ca_topic_score_codex":0.00959873,"about_ca_topic_score_gemma":0.010139022,"teacher_disagreement_score":0.047930658,"about_ca_system_score_codex":0.002134442,"about_ca_system_score_gemma":0.0027413722,"threshold_uncertainty_score":0.25348455},"labels":[],"label_agreement":null},{"id":"W1979991500","doi":"10.1007/s10664-012-9224-x","title":"Configuring latent Dirichlet allocation based feature location","year":2012,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":92,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"U.S. Department of Education; National Science Foundation","keywords":"Latent Dirichlet allocation; Computer science; Feature (linguistics); Source code; Heuristics; Context (archaeology); Java; Artificial intelligence; Topic model; Code (set theory); Measure (data warehouse); Data mining; Natural language processing; Information retrieval; Programming language","score_opus":0.024902158345490696,"score_gpt":0.2743756658225636,"score_spread":0.24947350747707292,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1979991500","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02798516,0.00018843859,0.9646391,0.00031779503,0.00012259265,0.00008263373,0.00032344516,0.00535415,0.0009866538],"genre_scores_gemma":[0.4894229,0.00012699878,0.503398,0.00036012544,0.00014429285,0.00040108917,0.0021226562,0.0009832985,0.0030406285],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9952041,0.0023403782,0.00027782223,0.0012115211,0.0005941085,0.0003720744],"domain_scores_gemma":[0.98939437,0.0063823764,0.00028595288,0.0023862347,0.0012302066,0.0003208873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046919524,0.0011722925,0.0019940569,0.001771715,0.0011988453,0.002288966,0.0035654048,0.0029960263,0.0059461007],"category_scores_gemma":[0.026178025,0.0011174115,0.0015627481,0.0020366644,0.0011304445,0.0043413946,0.0044274866,0.0025514,0.00408932],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028060828,0.0006807047,0.011383732,0.00021700388,0.00036026502,0.0003402931,0.00068906037,0.17105553,0.0198756,0.01983182,0.016937558,0.7558224],"study_design_scores_gemma":[0.00009839551,0.000056941408,0.00061625656,0.000011835741,0.00005144429,0.00008170229,0.000094595445,0.9660085,0.006740224,0.024545614,0.0016655826,0.000028918985],"about_ca_topic_score_codex":0.005126292,"about_ca_topic_score_gemma":0.0069454857,"teacher_disagreement_score":0.0059461007,"about_ca_system_score_codex":0.0012473048,"about_ca_system_score_gemma":0.0017349288,"threshold_uncertainty_score":0.024813771},"labels":[],"label_agreement":null},{"id":"W1980130600","doi":"10.1007/s10664-014-9329-5","title":"Special issue on program comprehension","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Program comprehension; Computer science; Comprehension; Data science; Psychology; Programming language; Software","score_opus":0.021811196545500226,"score_gpt":0.29366010915216767,"score_spread":0.27184891260666744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1980130600","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011580094,0.029325766,0.0035373315,0.0758656,0.77429914,0.00016459535,0.0012807695,0.000635164,0.11373373],"genre_scores_gemma":[0.0042727776,0.0167839,0.00096330454,0.010639122,0.6883703,0.00018668386,0.0022598014,0.0007136253,0.27581063],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980794,0.0003017076,0.00016433683,0.00034874014,0.00086265634,0.00024317861],"domain_scores_gemma":[0.9901154,0.0028528455,0.00058399583,0.0010536802,0.0031822075,0.0022118727],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026461454,0.0021313396,0.0025027762,0.006820924,0.0020791267,0.0068121497,0.0023601782,0.004863947,0.19952083],"category_scores_gemma":[0.0090997,0.0006864965,0.0013021879,0.004038854,0.0012781651,0.0058516664,0.0046474207,0.0045236517,0.07241065],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028572023,0.000045407713,0.00012330378,0.00017769364,0.00000795802,0.00003791085,0.000023462999,0.000043125478,0.00018575165,0.0016720854,0.9651267,0.03252801],"study_design_scores_gemma":[0.000016166101,0.000048420552,0.0010803014,0.00028672622,0.000015027721,0.00012879187,0.000045849512,0.0001321158,0.00016046806,0.004967036,0.9931076,0.0000114949125],"about_ca_topic_score_codex":0.00093493186,"about_ca_topic_score_gemma":0.0023611442,"teacher_disagreement_score":0.19952083,"about_ca_system_score_codex":0.001855172,"about_ca_system_score_gemma":0.0027264266,"threshold_uncertainty_score":0.66746366},"labels":[],"label_agreement":null},{"id":"W1981075560","doi":"10.1007/s10664-013-9274-8","title":"Studying the relationship between logging characteristics and the code quality of platform software","year":2013,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":83,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Software; Product metric; Process (computing); Relation (database); Source lines of code; Logging; Software quality; Code (set theory); Database; Quality (philosophy); Product (mathematics); Software development; Software engineering; Data mining; Operating system; Programming language; Set (abstract data type)","score_opus":0.1071977212306796,"score_gpt":0.3313187722431654,"score_spread":0.2241210510124858,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1981075560","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99929667,0.000029966688,0.00038427988,0.00002842717,9.790556e-7,0.0000026136277,0.000023450122,0.0000057768993,0.00022774888],"genre_scores_gemma":[0.99960035,0.000018083621,0.00020681089,0.000004415531,0.0000021831586,0.0000018234712,0.000052806656,0.0000041307358,0.00010935617],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99800843,0.0006725447,0.00019854162,0.00030302146,0.0005433173,0.00027406792],"domain_scores_gemma":[0.8040397,0.14255577,0.037680328,0.005471352,0.007526554,0.002726274],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032256271,0.00026962408,0.00016723873,0.0017237697,0.0004061348,0.0015063495,0.0005910607,0.00068484363,0.002097047],"category_scores_gemma":[0.067163475,0.000351052,0.00037905216,0.0026370964,0.00074996264,0.0023992267,0.00074343185,0.0013404033,0.00028787873],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012693941,0.00017590798,0.99280506,0.000015329546,0.00006362877,0.000043674267,0.00023187899,0.0009770009,0.00082951004,0.00024090018,0.000039454837,0.0044508306],"study_design_scores_gemma":[0.000008034489,0.00024231337,0.98972356,0.000007778283,0.000053543165,0.0000952793,0.0005756542,0.007289979,0.0013805238,0.00046945593,0.00014192717,0.000011943666],"about_ca_topic_score_codex":0.0048947893,"about_ca_topic_score_gemma":0.008909918,"teacher_disagreement_score":0.0048947893,"about_ca_system_score_codex":0.00074850395,"about_ca_system_score_gemma":0.001101173,"threshold_uncertainty_score":0.017058909},"labels":[],"label_agreement":null},{"id":"W1981621595","doi":"10.1007/s10664-010-9132-x","title":"Testing the theory of relative defect proneness for closed-source software","year":2010,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Children's Hospital of Eastern Ontario","funders":"","keywords":"Open source software; Software; Open source; Computer science; Software engineering; Data science; Programming language","score_opus":0.0349592160170909,"score_gpt":0.28196653910387043,"score_spread":0.24700732308677953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1981621595","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98313934,0.00006072074,0.013686033,0.00033224517,0.000021416785,0.000034052933,0.00010809808,0.000053048414,0.002565046],"genre_scores_gemma":[0.9979159,0.000017439092,0.0017329851,0.000034271277,0.000015541356,0.000028933107,0.0001139423,0.000012618103,0.00012827308],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.97975576,0.011212095,0.0010048938,0.003900984,0.0033391921,0.0007871195],"domain_scores_gemma":[0.34976456,0.60609597,0.019355701,0.01574102,0.00684461,0.002198206],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03089593,0.0011432688,0.00071107683,0.0025275634,0.0007718414,0.0019869825,0.0028777295,0.0022870654,0.006203123],"category_scores_gemma":[0.26443797,0.00044491273,0.0013975863,0.0017017145,0.0046493914,0.006115169,0.0023542396,0.0024349678,0.00045014624],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003945046,0.003972724,0.7239221,0.0006484842,0.0023822344,0.0005690422,0.0038523658,0.058184784,0.0074898987,0.11041311,0.0018536262,0.082766585],"study_design_scores_gemma":[0.001019995,0.008686923,0.4147912,0.00017311831,0.0007627332,0.0010283676,0.0033891576,0.38493633,0.008474749,0.17514142,0.0014425487,0.00015340424],"about_ca_topic_score_codex":0.0012065943,"about_ca_topic_score_gemma":0.0006760867,"teacher_disagreement_score":0.96910405,"about_ca_system_score_codex":0.0011934493,"about_ca_system_score_gemma":0.0013478087,"threshold_uncertainty_score":0.16339529},"labels":[],"label_agreement":null},{"id":"W1982036946","doi":"10.1007/s10664-013-9260-1","title":"An experimental investigation on the effects of context on source code identifiers splitting and expansion","year":2013,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Identifier; Computer science; Source code; Program comprehension; Context (archaeology); Documentation; Acronym; Unique identifier; Internal documentation; Code (set theory); Set (abstract data type); Software documentation; World Wide Web; Information retrieval; Software; Programming language; Software system; Software development; Software development process; Linguistics; Software construction","score_opus":0.021987243302096875,"score_gpt":0.26804226194432085,"score_spread":0.24605501864222398,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982036946","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9935688,0.00009604885,0.002399772,0.00010630603,0.000039476414,0.00021205227,0.00017649347,0.00010338368,0.00329773],"genre_scores_gemma":[0.99205387,0.0000791648,0.0054283463,0.00012401782,0.000036152287,0.00043746462,0.00017511865,0.00011414025,0.0015517663],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.99522734,0.0020110977,0.0005567815,0.0012007197,0.0007560964,0.00024802392],"domain_scores_gemma":[0.78182954,0.18394509,0.012187157,0.015128563,0.0043476634,0.0025619802],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004278875,0.000783817,0.00055523013,0.0004662779,0.00093963987,0.0016096131,0.0013180306,0.001092048,0.01003117],"category_scores_gemma":[0.09147315,0.00091106637,0.00029487276,0.0005689462,0.0015758759,0.0026428918,0.0019973086,0.0018062802,0.0007759554],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.07506744,0.082536645,0.1170835,0.0031804345,0.00043632014,0.0010954902,0.028453806,0.007210187,0.4887811,0.01215982,0.0032278828,0.18076737],"study_design_scores_gemma":[0.010762077,0.107458584,0.47916186,0.00055540795,0.0018035788,0.0024371955,0.01644204,0.04659492,0.2815251,0.031836245,0.020695368,0.0007276614],"about_ca_topic_score_codex":0.000703515,"about_ca_topic_score_gemma":0.0009931343,"teacher_disagreement_score":0.01003117,"about_ca_system_score_codex":0.0004268356,"about_ca_system_score_gemma":0.00088787137,"threshold_uncertainty_score":0.033557594},"labels":[],"label_agreement":null},{"id":"W1985892950","doi":"10.1007/s10664-015-9371-y","title":"An empirical study of integration activities in distributions of open source software","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; Queen's University; Polytechnique Montréal","funders":"","keywords":"Computer science; Component (thermodynamics); Reuse; System integration; Software engineering; Software; Context (archaeology); Software quality; Component-based software engineering; Software development; Systems engineering; Engineering; Operating system","score_opus":0.06030563834253202,"score_gpt":0.3592406911326763,"score_spread":0.2989350527901443,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1985892950","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9986883,0.000030548123,0.0005973058,0.000036575755,0.0000011094274,0.000007749917,0.000023387818,0.000008472551,0.00060661946],"genre_scores_gemma":[0.999102,0.000023767665,0.00053677184,0.0000062789854,0.000004615678,0.000006629161,0.00006471445,0.000006426392,0.0002487975],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9966625,0.0017515981,0.00017971275,0.00031392806,0.0007977767,0.00029436577],"domain_scores_gemma":[0.8242804,0.1320251,0.021244299,0.0070063053,0.011331908,0.004112005],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004107342,0.0002463054,0.00020382751,0.0026957057,0.00089448906,0.0013958,0.0007967394,0.00068318,0.0023127813],"category_scores_gemma":[0.07991409,0.00025712384,0.0002297931,0.0031578674,0.0012797603,0.0034552964,0.0014599416,0.00128049,0.00043123012],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038512133,0.0014157632,0.9459757,0.000048231595,0.000037274218,0.00027604395,0.0076936064,0.0011501141,0.0019877688,0.00315874,0.00036933448,0.037502356],"study_design_scores_gemma":[0.000038689708,0.00061350607,0.96426946,0.000037921847,0.000040693205,0.00065065554,0.011336148,0.016332874,0.0018870295,0.003173647,0.0015895324,0.000029893781],"about_ca_topic_score_codex":0.0039833426,"about_ca_topic_score_gemma":0.004832301,"teacher_disagreement_score":0.004107342,"about_ca_system_score_codex":0.0011044188,"about_ca_system_score_gemma":0.00087317626,"threshold_uncertainty_score":0.02172196},"labels":[],"label_agreement":null},{"id":"W1987409374","doi":"10.1007/s10664-008-9084-6","title":"Another viewpoint on “evaluating web software reliability based on workload and failure data extracted from server logs”","year":2008,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Reliability and Analysis Research","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Workload; Reliability (semiconductor); Computer science; Replication (statistics); Reliability engineering; Software quality; Web application; Term (time); Data mining; Software; World Wide Web; Engineering; Statistics; Software development; Operating system","score_opus":0.08123406468878247,"score_gpt":0.3199451369208011,"score_spread":0.23871107223201865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1987409374","genre_codex":"methods","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.094885044,0.0038616285,0.7297541,0.08350812,0.0023272801,0.00032373017,0.0006006214,0.0011427869,0.08359671],"genre_scores_gemma":[0.84421617,0.0016376178,0.11260908,0.022031976,0.003024719,0.0004488438,0.0005082857,0.00027916083,0.015244239],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9887565,0.0040346817,0.0007140467,0.0017721753,0.0042247307,0.0004977975],"domain_scores_gemma":[0.94326174,0.029120052,0.0037399353,0.0078216065,0.014567713,0.0014889309],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012006881,0.0020566229,0.0018326612,0.006280961,0.0016257508,0.00602912,0.0034778987,0.0061269407,0.005110892],"category_scores_gemma":[0.056660756,0.00079893315,0.0017864864,0.0034571162,0.007531749,0.0067669908,0.0025129535,0.00602232,0.0017412449],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079046714,0.0012775937,0.051051695,0.0012987726,0.00091918523,0.0012017116,0.0059172767,0.020081257,0.015489518,0.6504656,0.04028566,0.21122125],"study_design_scores_gemma":[0.00060045184,0.0030161806,0.063037395,0.0014001402,0.0011466421,0.0020745273,0.00717251,0.119777545,0.020291371,0.71244425,0.06852041,0.0005185957],"about_ca_topic_score_codex":0.0035021687,"about_ca_topic_score_gemma":0.0033277425,"teacher_disagreement_score":0.012006881,"about_ca_system_score_codex":0.002732167,"about_ca_system_score_gemma":0.0022460213,"threshold_uncertainty_score":0.06349921},"labels":[],"label_agreement":null},{"id":"W1987810861","doi":"10.1007/s10664-005-1288-4","title":"Requirements Engineering and Downstream Software Development: Findings from a Case Study","year":2005,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":53,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"University of Victoria","keywords":"Software project management; Software development process; Software development; Software engineering; Computer science; Requirements analysis; Requirement prioritization; Software requirements; Requirements engineering; Process management; Requirement; Documentation; Personal software process; Team software process; Systems engineering; Engineering; Software construction; Software","score_opus":0.030852546566778504,"score_gpt":0.28626038873216897,"score_spread":0.2554078421653905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1987810861","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9880046,0.0001065009,0.0032089439,0.0004213556,0.0000061033775,0.00010848777,0.000020789781,0.000020494868,0.00810267],"genre_scores_gemma":[0.9937975,0.00022506002,0.0031258846,0.00014687161,0.000006121209,0.00007660861,0.000038152917,0.000019092466,0.0025646836],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9917726,0.004970929,0.00034853903,0.00043242314,0.0019200989,0.0005553659],"domain_scores_gemma":[0.9041737,0.08122223,0.0042732377,0.003163552,0.005345772,0.0018214977],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007246606,0.00037522233,0.00033658743,0.0015775235,0.0038290662,0.0017461635,0.001341531,0.0017581861,0.002734946],"category_scores_gemma":[0.03991626,0.00045988662,0.0003653099,0.0015541273,0.0018288941,0.0022621406,0.0021272078,0.0024393594,0.00042812782],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012178697,0.027948366,0.2593118,0.0012675985,0.00010126847,0.029219665,0.35696805,0.0041951244,0.015101195,0.022931503,0.005562255,0.2761753],"study_design_scores_gemma":[0.0004449867,0.009008072,0.28411663,0.0015241813,0.00033124583,0.033557966,0.544664,0.021141777,0.04182614,0.010471448,0.05258059,0.0003330097],"about_ca_topic_score_codex":0.00427292,"about_ca_topic_score_gemma":0.010625629,"teacher_disagreement_score":0.007246606,"about_ca_system_score_codex":0.0021351767,"about_ca_system_score_gemma":0.0030831147,"threshold_uncertainty_score":0.038324118},"labels":[],"label_agreement":null},{"id":"W1987963388","doi":"10.1007/s10664-015-9370-z","title":"Studying high impact fix-inducing changes","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Change impact analysis; Computer science; Software; Quality (philosophy); Risk analysis (engineering); Data science; Software engineering","score_opus":0.07085915187211694,"score_gpt":0.32367575130540127,"score_spread":0.25281659943328433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1987963388","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9881996,0.00025719873,0.0064390474,0.00010156892,0.000020185344,0.000031275777,0.00026321015,0.0001163987,0.0045715407],"genre_scores_gemma":[0.99571383,0.00012276722,0.0024044267,0.00003108743,0.000011490201,0.000014792356,0.00037883205,0.000036408514,0.0012863771],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99868566,0.00035253353,0.00007328843,0.00026375143,0.00043148806,0.00019324181],"domain_scores_gemma":[0.965533,0.023380185,0.003820332,0.004294248,0.0022710373,0.0007011982],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0015792164,0.0003409708,0.00031384887,0.001168645,0.00048695464,0.00081095315,0.0006311093,0.00070160185,0.0062878877],"category_scores_gemma":[0.026664274,0.00023424345,0.00030807007,0.0014109147,0.00050160347,0.0011456001,0.00055331684,0.0011224283,0.000554653],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002174024,0.0049808677,0.46742672,0.0012868122,0.00063613534,0.0019973668,0.00191822,0.05183289,0.0944741,0.05436312,0.007267329,0.31164238],"study_design_scores_gemma":[0.00028711974,0.003504778,0.72352034,0.00022734515,0.0004977784,0.0026736087,0.0034994094,0.11771883,0.08268793,0.04907685,0.016188268,0.000117732154],"about_ca_topic_score_codex":0.0010804763,"about_ca_topic_score_gemma":0.0020876564,"teacher_disagreement_score":0.9984208,"about_ca_system_score_codex":0.00056182913,"about_ca_system_score_gemma":0.00048284561,"threshold_uncertainty_score":0.021035135},"labels":[],"label_agreement":null},{"id":"W1991263523","doi":"10.1007/s10664-009-9110-3","title":"Guest editors introduction: special issue on mining software repositories","year":2009,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"European Metrology Programme for Innovation and Research","keywords":"Computer science; Software engineering; Data science; World Wide Web","score_opus":0.010820047409559233,"score_gpt":0.25210937070345196,"score_spread":0.24128932329389274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1991263523","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00045453454,0.0074629127,0.0028807605,0.026004761,0.95426404,0.000058868027,0.0004035886,0.0004639744,0.008006459],"genre_scores_gemma":[0.002245732,0.006553637,0.0014017546,0.0067590284,0.9222661,0.000056464476,0.00065735204,0.00048210987,0.059577834],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99791104,0.0002763702,0.00026809334,0.00045647335,0.0009027547,0.00018517795],"domain_scores_gemma":[0.98350936,0.003057927,0.0013153397,0.000869219,0.007769425,0.0034787436],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003779316,0.0023250808,0.0027029738,0.006508915,0.0017188346,0.007010654,0.002161064,0.0032197484,0.080812156],"category_scores_gemma":[0.012247809,0.0008770917,0.001339563,0.0033141093,0.00068533816,0.005481078,0.0031442584,0.004880225,0.040135864],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028725985,0.000018334555,0.00012597951,0.00013256034,0.000009895297,0.00007192054,0.000009889287,0.000047354635,0.00019548707,0.00027850852,0.9787541,0.020327175],"study_design_scores_gemma":[0.000038284223,0.00009159163,0.0010495943,0.00018003046,0.000039400926,0.00035715345,0.000049923612,0.0004727038,0.00041097114,0.0017093752,0.995572,0.000028921691],"about_ca_topic_score_codex":0.00054485374,"about_ca_topic_score_gemma":0.002137423,"teacher_disagreement_score":0.080812156,"about_ca_system_score_codex":0.0009264868,"about_ca_system_score_gemma":0.0012325682,"threshold_uncertainty_score":0.2703436},"labels":[],"label_agreement":null},{"id":"W1991613282","doi":"10.1007/s10664-010-9150-8","title":"A field study of API learning obstacles","year":2010,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":352,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"McGill University; Microsoft Research","keywords":"Documentation; Computer science; Programmer; Presentation (obstetrics); Field (mathematics); World Wide Web; Software engineering; Software documentation; Internal documentation; Multimedia; Software development; Software; Programming language; Software development process","score_opus":0.017733432877838518,"score_gpt":0.2866929103016268,"score_spread":0.2689594774237883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1991613282","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9914458,0.000065053086,0.0017018118,0.00018560377,0.00001502878,0.00020139701,0.00010459795,0.00003871158,0.006242027],"genre_scores_gemma":[0.99280375,0.0001134372,0.001679601,0.0001119844,0.000012381041,0.00018487738,0.00017919597,0.000018074192,0.0048965234],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9978828,0.00092003145,0.00012410858,0.00032024586,0.00049273594,0.00026010018],"domain_scores_gemma":[0.9486915,0.034418553,0.0029169226,0.0030336631,0.007387905,0.0035515027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035845973,0.00039628908,0.00038244444,0.0024594048,0.0028134368,0.0014864256,0.0015115013,0.0011249878,0.0071304566],"category_scores_gemma":[0.026904583,0.00047759406,0.00029095422,0.001810293,0.0016359268,0.002756436,0.001628241,0.002297278,0.000967709],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0061581796,0.11965721,0.369503,0.0012044397,0.00009226475,0.0029006545,0.13451378,0.003056696,0.018081622,0.025637178,0.016447537,0.30274734],"study_design_scores_gemma":[0.0012575759,0.03595973,0.53708005,0.0008470285,0.00022087379,0.002551469,0.30097997,0.018330079,0.023863547,0.02203592,0.056562986,0.00031076482],"about_ca_topic_score_codex":0.006725766,"about_ca_topic_score_gemma":0.008847254,"teacher_disagreement_score":0.0071304566,"about_ca_system_score_codex":0.0017108708,"about_ca_system_score_gemma":0.003065803,"threshold_uncertainty_score":0.02385372},"labels":[],"label_agreement":null},{"id":"W1991842606","doi":"10.1007/s10664-014-9302-3","title":"Waste identification and elimination in information technology organizations","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Quality and Supply Management","field":"Business, Management and Accounting","cited_by":32,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Categorization; Identification (biology); Lean manufacturing; Knowledge management; Process management; Business; Engineering; Computer science; Operations management; Artificial intelligence","score_opus":0.007171616111908883,"score_gpt":0.21267061111200308,"score_spread":0.2054989950000942,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1991842606","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9954692,0.00019440586,0.0018421967,0.00019205835,0.000004913004,0.000030301659,0.000044966324,0.000009590241,0.0022123253],"genre_scores_gemma":[0.997771,0.000091772235,0.0007918108,0.000029592717,0.000003866608,0.000022958468,0.00006943345,0.000007242503,0.0012122199],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9913874,0.004397329,0.00058842957,0.00046124787,0.0017941887,0.0013714095],"domain_scores_gemma":[0.92481494,0.049821857,0.015286507,0.0025094652,0.006318784,0.0012485381],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011640442,0.00037938548,0.0007710757,0.0067475745,0.0022032452,0.0036423078,0.0013597116,0.001677079,0.002942484],"category_scores_gemma":[0.055785052,0.0004989792,0.0008173162,0.0071268347,0.0024376786,0.004685441,0.0027442987,0.001347135,0.0003235321],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010473526,0.0016439374,0.9372482,0.0001536356,0.00013011505,0.00013183439,0.0050622784,0.005824044,0.00026291236,0.009355491,0.00058414025,0.038556058],"study_design_scores_gemma":[0.000068476984,0.0012227296,0.8754525,0.00038618888,0.0003948064,0.00024082736,0.04857486,0.03933851,0.0040049762,0.027240714,0.0029927737,0.00008258609],"about_ca_topic_score_codex":0.027354322,"about_ca_topic_score_gemma":0.03462915,"teacher_disagreement_score":0.027354322,"about_ca_system_score_codex":0.0032443565,"about_ca_system_score_gemma":0.006304051,"threshold_uncertainty_score":0.061561286},"labels":[],"label_agreement":null},{"id":"W1991867644","doi":"10.1007/s10664-015-9375-7","title":"Analyzing and automatically labelling the types of user issues that are raised in mobile app reviews","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":188,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Queen's University","funders":"","keywords":"App store; Computer science; World Wide Web; Download; Internet privacy; Mobile apps; Mobile device; Analytics; Notice; Data science","score_opus":0.052240163040090756,"score_gpt":0.31900221406896706,"score_spread":0.2667620510288763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1991867644","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92182285,0.008429823,0.043867704,0.002083048,0.00072946347,0.0010159647,0.006626593,0.003083073,0.012341457],"genre_scores_gemma":[0.90672195,0.0024087566,0.073671125,0.0005298289,0.00040555844,0.00060249533,0.008159434,0.00039396717,0.0071069133],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.98891246,0.0035201486,0.0013476085,0.0011936752,0.0046751774,0.00035094388],"domain_scores_gemma":[0.8451639,0.108508475,0.0188138,0.0036064289,0.022743845,0.0011636401],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006229083,0.00080312986,0.0007328273,0.012319657,0.001039726,0.0027660078,0.0007881423,0.0014499133,0.0010666024],"category_scores_gemma":[0.08608788,0.0005188828,0.0006743261,0.0041393484,0.0003684386,0.002666769,0.0013023436,0.0010668656,0.0011805536],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011599872,0.000435178,0.360364,0.0075349472,0.00046871678,0.0028703157,0.012884311,0.0013878444,0.056433152,0.0029637518,0.043104563,0.51039326],"study_design_scores_gemma":[0.000116610354,0.0011124874,0.6913583,0.0031181115,0.0014892172,0.008753801,0.013747976,0.07616106,0.06316012,0.0056732856,0.13487162,0.00043739166],"about_ca_topic_score_codex":0.0037181955,"about_ca_topic_score_gemma":0.01187222,"teacher_disagreement_score":0.012319657,"about_ca_system_score_codex":0.00069938286,"about_ca_system_score_gemma":0.0019277618,"threshold_uncertainty_score":0.03294295},"labels":[],"label_agreement":null},{"id":"W1996342119","doi":"10.1007/s10664-011-9188-2","title":"Introduction to the special issue on software repository mining in 2009","year":2011,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Software engineering; Software; Data science; Operating system","score_opus":0.022772899001632055,"score_gpt":0.257169182466531,"score_spread":0.23439628346489894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1996342119","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012208449,0.053963155,0.051368836,0.09568327,0.75634664,0.0003228231,0.0035437685,0.001826418,0.035724275],"genre_scores_gemma":[0.0062692664,0.051010184,0.029010285,0.038056057,0.5995164,0.0002373334,0.0078927595,0.0022750092,0.2657328],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9961427,0.00062309636,0.00054807856,0.0008314811,0.0016337732,0.00022093348],"domain_scores_gemma":[0.97921705,0.0059064445,0.0011497589,0.0018744979,0.0094211735,0.0024309333],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062371315,0.0016737395,0.0026281804,0.008090917,0.0016585254,0.0073762364,0.0019392071,0.0029470983,0.056955993],"category_scores_gemma":[0.019519234,0.0009044778,0.0015214063,0.0060731196,0.0012312909,0.008996458,0.003911788,0.0057556983,0.039003015],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026684367,0.00003309595,0.00021017806,0.00025041803,0.000015005678,0.000050467028,0.00002573114,0.00014931754,0.00042108112,0.0015336248,0.9122657,0.08501866],"study_design_scores_gemma":[0.000007385538,0.00003726235,0.0005962593,0.00020937469,0.000012852863,0.00020763105,0.000028770179,0.00044569751,0.0003333879,0.0026455228,0.99545175,0.000024084346],"about_ca_topic_score_codex":0.0019072471,"about_ca_topic_score_gemma":0.005799961,"teacher_disagreement_score":0.056955993,"about_ca_system_score_codex":0.0024077077,"about_ca_system_score_gemma":0.003403722,"threshold_uncertainty_score":0.1905368},"labels":[],"label_agreement":null},{"id":"W1998785137","doi":"10.1007/s10664-008-9063-y","title":"Triangulation as a basis for knowledge discovery in software engineering","year":2008,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Artificial Intelligence in Education","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Software engineering; Triangulation; Basis (linear algebra); Data science; Cartography; Mathematics; Geography","score_opus":0.05327967590589569,"score_gpt":0.3145633665390798,"score_spread":0.26128369063318413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1998785137","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017131101,0.0028267961,0.9569708,0.003918294,0.00024597568,0.0009080333,0.00055579323,0.00021431742,0.017228898],"genre_scores_gemma":[0.3161525,0.0016038818,0.67814493,0.00042706088,0.00013873736,0.0012646375,0.00091059017,0.00007518558,0.0012825022],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.87062484,0.10066441,0.007153795,0.0069587566,0.013396174,0.0012020052],"domain_scores_gemma":[0.6381206,0.3010013,0.009000066,0.037726432,0.012653153,0.0014984738],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.051699013,0.0010592968,0.0029864302,0.015149846,0.0056119678,0.009385525,0.0065645943,0.003848888,0.0060496465],"category_scores_gemma":[0.3189736,0.0012879837,0.0031147937,0.013662842,0.013341539,0.0141292205,0.012048295,0.005469871,0.0010376544],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027538865,0.00013667966,0.008570416,0.0022868558,0.00076027313,0.00046262355,0.014389674,0.01069996,0.00061041326,0.7588676,0.003801402,0.19913864],"study_design_scores_gemma":[0.000057378762,0.000051407656,0.0009003866,0.00075358973,0.00012158755,0.00027081987,0.0024135015,0.021450391,0.00052111316,0.9644538,0.008950888,0.00005509822],"about_ca_topic_score_codex":0.01098168,"about_ca_topic_score_gemma":0.010949287,"teacher_disagreement_score":0.94830096,"about_ca_system_score_codex":0.0043488992,"about_ca_system_score_gemma":0.009785904,"threshold_uncertainty_score":0.27341378},"labels":[],"label_agreement":null},{"id":"W2002641269","doi":"10.1007/s10664-014-9315-y","title":"An empirical study on the importance of source code entities for requirements traceability","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Waterloo","funders":"","keywords":"Computer science; Information retrieval; Traceability; Source code; Weighting; Search engine indexing; Rank (graph theory); Data mining; Software engineering; Programming language","score_opus":0.05999686482834988,"score_gpt":0.3489345172867725,"score_spread":0.2889376524584226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002641269","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9968644,0.00007780567,0.0013146653,0.00011237478,0.000003660893,0.000030697072,0.00003662547,0.000008162946,0.0015515782],"genre_scores_gemma":[0.9986204,0.00004920134,0.0010145019,0.000020140626,0.0000036487654,0.000013528911,0.00006786043,0.000006771797,0.00020391987],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99054915,0.0052949777,0.00071197166,0.0007286245,0.0023461604,0.00036912464],"domain_scores_gemma":[0.3697808,0.5703203,0.029988302,0.011228925,0.015800359,0.0028813074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013134783,0.00031313722,0.00017528802,0.0030472584,0.00080040656,0.0014995472,0.0009260275,0.0008791077,0.002844926],"category_scores_gemma":[0.21636586,0.00027645135,0.00032293488,0.0030462793,0.0015832394,0.004169205,0.0011633161,0.001961907,0.00024936374],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007239827,0.0029902891,0.9129549,0.0003344015,0.00009686486,0.00034721536,0.009467962,0.0012336025,0.0036939266,0.0036830816,0.00037379737,0.06409977],"study_design_scores_gemma":[0.00007548805,0.0012600981,0.96458125,0.00017065475,0.00017558095,0.00062030944,0.013458465,0.010476842,0.004544277,0.0021485623,0.002455292,0.000033258948],"about_ca_topic_score_codex":0.0033265215,"about_ca_topic_score_gemma":0.0049050865,"teacher_disagreement_score":0.013134783,"about_ca_system_score_codex":0.0011256079,"about_ca_system_score_gemma":0.001974818,"threshold_uncertainty_score":0.06946415},"labels":[],"label_agreement":null},{"id":"W2013851052","doi":"10.1007/s10664-006-9031-3","title":"A practical approach to testing GUI systems","year":2006,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Component (thermodynamics); Set (abstract data type); Graph; Graphical user interface; Software engineering; Programming language; Theoretical computer science","score_opus":0.060666448574408956,"score_gpt":0.297102360528753,"score_spread":0.23643591195434405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2013851052","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004894258,0.00016470095,0.9773457,0.003998331,0.00007409487,0.00032064237,0.00006687469,0.0009962765,0.012139055],"genre_scores_gemma":[0.100347,0.00030690272,0.88969845,0.00074597035,0.000099883706,0.00069613894,0.00013973871,0.00014192302,0.007823979],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.987906,0.0064177066,0.00046127307,0.00095943204,0.0039385096,0.00031707613],"domain_scores_gemma":[0.96282357,0.0219869,0.0012269198,0.008175828,0.004870316,0.00091632834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00861635,0.0017118378,0.0009752092,0.0028573736,0.0019701368,0.003348725,0.004666108,0.003875441,0.021351121],"category_scores_gemma":[0.054907292,0.0012181255,0.000702248,0.0020249747,0.0037731796,0.0063316994,0.0050050262,0.0054374337,0.0042239535],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016457739,0.0012773499,0.0059736646,0.0006886883,0.00008033001,0.0008401395,0.0015688271,0.01230979,0.008586078,0.3409224,0.018601198,0.60898703],"study_design_scores_gemma":[0.0002987284,0.0008661661,0.004537025,0.00041697483,0.00008093722,0.0032851151,0.0019382264,0.09948977,0.0061980607,0.818778,0.06401217,0.00009884553],"about_ca_topic_score_codex":0.0014967447,"about_ca_topic_score_gemma":0.003424928,"teacher_disagreement_score":0.021351121,"about_ca_system_score_codex":0.0011344921,"about_ca_system_score_gemma":0.0026466008,"threshold_uncertainty_score":0.07142663},"labels":[],"label_agreement":null},{"id":"W2019257047","doi":"10.1007/s10664-015-9381-9","title":"An empirical study of the impact of modern code review practices on software quality","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":320,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Queen's University","funders":"","keywords":"Code review; Software quality; Computer science; Software engineering; Software quality analyst; Software inspection; Software peer review; Software quality management; Static program analysis; Software construction; Software quality assurance; Software development; Software; Programming language","score_opus":0.14670699522306,"score_gpt":0.4503907650299953,"score_spread":0.30368376980693534,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2019257047","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9981111,0.00033354427,0.00028904053,0.00018904688,0.0000063972707,0.000025702784,0.00003518358,0.00001271137,0.0009971929],"genre_scores_gemma":[0.99908066,0.00013081003,0.00047691743,0.00004606325,0.000014801438,0.000014533256,0.00003727883,0.0000051808456,0.00019381648],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9840423,0.007693855,0.0011451403,0.0011252231,0.005061395,0.00093203847],"domain_scores_gemma":[0.42364755,0.4192705,0.10102864,0.013889635,0.03242593,0.009737784],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.014670518,0.0002526799,0.0002813511,0.0031193113,0.0009291299,0.0018738813,0.0010626176,0.0009039212,0.0019077983],"category_scores_gemma":[0.171371,0.00037137713,0.0004334131,0.0032487742,0.0017555343,0.002449424,0.001432291,0.0018614616,0.00020268641],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013689232,0.006450636,0.8537005,0.00059933175,0.0003711723,0.0002747418,0.005974135,0.0022331676,0.0037640866,0.0018951807,0.0011070912,0.122261055],"study_design_scores_gemma":[0.0001277591,0.0031404078,0.9865919,0.00015017923,0.00017759226,0.00021836744,0.0034837709,0.0027701317,0.0013104846,0.0005239119,0.0014712424,0.000034276116],"about_ca_topic_score_codex":0.0075919926,"about_ca_topic_score_gemma":0.013931158,"teacher_disagreement_score":0.9853295,"about_ca_system_score_codex":0.003458273,"about_ca_system_score_gemma":0.0042577637,"threshold_uncertainty_score":0.077586055},"labels":[],"label_agreement":null},{"id":"W2020503784","doi":"10.1007/s10664-014-9340-x","title":"The kanban approach, between agility and leanness: a systematic review","year":2014,"lang":"en","type":"review","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":64,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Kanban; Agile software development; Variety (cybernetics); Field (mathematics); Process management; Scrum; Computer science; Management science; Lean software development; Product (mathematics); Software; Software development; Engineering; Knowledge management; Software development process; Software engineering; Control (management); Artificial intelligence; Mathematics","score_opus":0.04499939975791214,"score_gpt":0.3342892531106428,"score_spread":0.28928985335273066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2020503784","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006332084,0.99835,0.00037563435,0.00022543478,0.00004911485,0.00011747696,0.00005349625,0.0000029851587,0.00019270324],"genre_scores_gemma":[0.010582231,0.98720586,0.0014646667,0.00038038887,0.00004074041,0.00018075873,0.00006566944,0.0000041023522,0.000075659205],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.98815876,0.0041827126,0.0036881675,0.0009592502,0.0028074,0.00020382619],"domain_scores_gemma":[0.95351386,0.0368966,0.005449906,0.00056413567,0.003262166,0.00031332174],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013674877,0.0015569074,0.005391953,0.011757115,0.00069996226,0.0035785865,0.0024207942,0.0020027203,0.0022833385],"category_scores_gemma":[0.053218488,0.0011366593,0.004063411,0.012435802,0.0019053248,0.004999935,0.002192566,0.0021385103,0.00024002166],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002476487,0.00008031079,0.0018907422,0.7120995,0.0061598914,0.00021143543,0.000816505,0.00031112728,0.00026056802,0.0016918489,0.0023067784,0.2739237],"study_design_scores_gemma":[0.0003348992,0.0003858079,0.007036287,0.89068353,0.047580328,0.0010951901,0.0020939286,0.00027776274,0.00048060887,0.0030544682,0.04685332,0.00012394747],"about_ca_topic_score_codex":0.006298336,"about_ca_topic_score_gemma":0.018868199,"teacher_disagreement_score":0.013674877,"about_ca_system_score_codex":0.0036842998,"about_ca_system_score_gemma":0.01512665,"threshold_uncertainty_score":0.07232046},"labels":[],"label_agreement":null},{"id":"W2023617068","doi":"10.1007/s10664-009-9125-9","title":"An empirical study on the efficiency of different design pattern representations in UML class diagrams","year":2010,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Université de Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Applications of UML; Unified Modeling Language; Class diagram; UML tool; Notation; Program comprehension; Class (philosophy); Software design pattern; Visualization; Software engineering; Programming language; Software; Data mining; Artificial intelligence; Software system","score_opus":0.04501694003319926,"score_gpt":0.33526054527152854,"score_spread":0.29024360523832926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2023617068","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98876816,0.00029305616,0.009249814,0.00015017917,0.000008245389,0.00008897389,0.00016427097,0.00010791076,0.0011694058],"genre_scores_gemma":[0.9739378,0.00024499834,0.024437761,0.00004540674,0.000009889386,0.000082010636,0.0005194106,0.00011880019,0.00060401316],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.96737176,0.0220432,0.0034930452,0.0020838184,0.0044634542,0.00054472155],"domain_scores_gemma":[0.26298916,0.6911853,0.015875394,0.017589066,0.011536772,0.00082426314],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020279475,0.00059135346,0.0005523052,0.004107081,0.0006377702,0.0024850126,0.0012688441,0.0015012102,0.0018495983],"category_scores_gemma":[0.29988626,0.00049330207,0.00086996524,0.004410884,0.0011293422,0.0061036455,0.0012342698,0.0013123576,0.00041965992],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004695905,0.00604865,0.33100826,0.0020225325,0.00059599313,0.00045678357,0.02157659,0.019780736,0.023294758,0.0073039704,0.0017128526,0.58150285],"study_design_scores_gemma":[0.0014574697,0.010052675,0.5982939,0.0009289977,0.0019796195,0.0026957996,0.022569787,0.28812686,0.048190173,0.011944801,0.013413497,0.00034648934],"about_ca_topic_score_codex":0.0020949466,"about_ca_topic_score_gemma":0.0027990923,"teacher_disagreement_score":0.020279475,"about_ca_system_score_codex":0.0013698824,"about_ca_system_score_gemma":0.0010306141,"threshold_uncertainty_score":0.10724944},"labels":[],"label_agreement":null},{"id":"W2024695825","doi":"10.1007/s10664-011-9163-y","title":"Qualitative research in software engineering","year":2011,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":59,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Qualitative research; Phenomenon; Ethnography; Context (archaeology); Social phenomenon; Social research; Participant observation; Grounded theory; Knowledge management; Software; Data science; Management science; Sociology; Qualitative property; Computer science; Epistemology; Engineering ethics; Social science; Engineering","score_opus":0.19842369192862866,"score_gpt":0.4223873846295447,"score_spread":0.22396369270091607,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2024695825","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.217332,0.013626576,0.29470983,0.07199603,0.0015543542,0.0060259947,0.0018531522,0.00021609222,0.39268598],"genre_scores_gemma":[0.9140066,0.0033146208,0.043064762,0.0061399844,0.00009640473,0.0046408023,0.0003011167,0.00009680623,0.028338907],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9370529,0.055093925,0.0008564559,0.0011205202,0.0048263436,0.0010499202],"domain_scores_gemma":[0.8056062,0.17010708,0.0034622457,0.004528739,0.013665453,0.0026303732],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.048521698,0.00047567155,0.00063066615,0.0027378,0.0061306152,0.005181123,0.0018876087,0.0016606696,0.011945734],"category_scores_gemma":[0.09610756,0.00051718863,0.000357796,0.0028917568,0.013185265,0.0051966547,0.0043473183,0.002424273,0.0011797453],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012631238,0.0002979247,0.0046508987,0.0034532908,0.000026008627,0.00035260804,0.31227863,0.0006564006,0.0020137422,0.5894342,0.009687048,0.0770229],"study_design_scores_gemma":[0.00013778711,0.00027199576,0.0046967464,0.005490239,0.000032261458,0.0005447264,0.4536624,0.001308046,0.0037249885,0.3390623,0.1910088,0.000059747206],"about_ca_topic_score_codex":0.006340111,"about_ca_topic_score_gemma":0.008989803,"teacher_disagreement_score":0.9514783,"about_ca_system_score_codex":0.009027703,"about_ca_system_score_gemma":0.013176179,"threshold_uncertainty_score":0.25661033},"labels":[],"label_agreement":null},{"id":"W2026170849","doi":"10.1007/s10664-014-9338-4","title":"On rapid releases and software testing: a case study and a semi-systematic literature review","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":113,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Context (archaeology); Test suite; Systematic review; Scope (computer science); Software engineering; Software; Software bug; Software testing; Test case; Operating system","score_opus":0.027427958343474917,"score_gpt":0.2857589749436667,"score_spread":0.25833101660019175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2026170849","genre_codex":"empirical","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46784762,0.4406667,0.024843963,0.010483641,0.00060904433,0.00710003,0.0029293322,0.00016670446,0.045353007],"genre_scores_gemma":[0.65670973,0.30823013,0.02434525,0.002848104,0.00023357927,0.002498043,0.0016678493,0.0000873259,0.0033799873],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.9768734,0.011182297,0.0041964217,0.0011131867,0.0058882358,0.0007464995],"domain_scores_gemma":[0.7839958,0.18809272,0.011231318,0.003712899,0.01197508,0.000992185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023031792,0.00061496143,0.0011464757,0.015959714,0.0016662847,0.0024295892,0.0015555552,0.0017761767,0.002354048],"category_scores_gemma":[0.06054345,0.00053625624,0.0010352304,0.0151587995,0.0018755641,0.0037718443,0.0026426166,0.0011660063,0.0003995851],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004212466,0.0008285102,0.024407139,0.14233932,0.0006001983,0.015050514,0.08042626,0.0013757391,0.0072703687,0.011337614,0.0106318565,0.7053112],"study_design_scores_gemma":[0.00019318188,0.0020177546,0.094265185,0.36184445,0.0028993608,0.014482143,0.20012529,0.001517544,0.011379422,0.008958283,0.30197826,0.00033918786],"about_ca_topic_score_codex":0.004378614,"about_ca_topic_score_gemma":0.013707575,"teacher_disagreement_score":0.023031792,"about_ca_system_score_codex":0.003465379,"about_ca_system_score_gemma":0.015954584,"threshold_uncertainty_score":0.12180519},"labels":[],"label_agreement":null},{"id":"W2031459347","doi":"10.1007/s10664-010-9152-6","title":"Using grounded theory to study the experience of software development","year":2011,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":236,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Grounded theory; Computer science; Context (archaeology); Software; Management science; Development theory; Qualitative research; Engineering; Sociology; Social science","score_opus":0.1170929886314408,"score_gpt":0.32886256978482953,"score_spread":0.21176958115338873,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2031459347","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.87723583,0.0014579562,0.062865414,0.005354784,0.00010777019,0.00053200003,0.00013048493,0.000053581505,0.052262083],"genre_scores_gemma":[0.9880185,0.0004715442,0.009718059,0.00034584856,0.0000070538836,0.00018021099,0.000055711986,0.000022440576,0.0011805391],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9752279,0.020396804,0.00044202918,0.0005796641,0.0025366347,0.0008170716],"domain_scores_gemma":[0.9000506,0.09212893,0.0016956422,0.002158588,0.002551999,0.0014142694],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015167295,0.0004794311,0.000485973,0.0031851318,0.003806886,0.0070222314,0.0018939989,0.0016908161,0.0017468142],"category_scores_gemma":[0.045189,0.0004927095,0.00036446686,0.00306313,0.013468383,0.008110503,0.0054661813,0.0040980605,0.0001739036],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000071008944,0.00031517004,0.0071290545,0.00043291,0.000033196968,0.00037644745,0.8756862,0.0008607789,0.0009625224,0.077960566,0.0008139543,0.03535817],"study_design_scores_gemma":[0.0000846797,0.0002893808,0.0075150914,0.0011121657,0.000036899462,0.0004272148,0.86641884,0.003772938,0.001621001,0.095361955,0.023297826,0.0000619143],"about_ca_topic_score_codex":0.006005622,"about_ca_topic_score_gemma":0.0075050024,"teacher_disagreement_score":0.015167295,"about_ca_system_score_codex":0.006744926,"about_ca_system_score_gemma":0.006614895,"threshold_uncertainty_score":0.08021331},"labels":[],"label_agreement":null},{"id":"W2034314383","doi":"10.1007/s10664-012-9212-1","title":"Preface to the special issue on program comprehension","year":2012,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Program comprehension; Computer science; Comprehension; Programming language; Software","score_opus":0.029704598744702493,"score_gpt":0.29145600796233845,"score_spread":0.26175140921763596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034314383","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00017587299,0.012084561,0.0011964549,0.04637506,0.9262857,0.00005872508,0.0004858592,0.0001605293,0.013177312],"genre_scores_gemma":[0.0011813324,0.0094217,0.0005813604,0.0132806515,0.8988694,0.00011083475,0.0010350659,0.0003517534,0.07516797],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99849033,0.0002450727,0.00020326312,0.00025646,0.00066793826,0.00013677057],"domain_scores_gemma":[0.97829205,0.006476304,0.0010241251,0.0011383432,0.009919528,0.0031496666],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032220904,0.0022392888,0.0027479343,0.006767408,0.002341315,0.0060141734,0.0022280037,0.0041215057,0.12590815],"category_scores_gemma":[0.016907461,0.00069236587,0.0013839193,0.0039975434,0.00094165996,0.004958136,0.003177568,0.00784356,0.065093644],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012209194,0.000017172626,0.000037378555,0.000078743615,0.0000032803675,0.000015150917,0.000012807162,0.000018278528,0.000059856444,0.00025823448,0.9883128,0.011174086],"study_design_scores_gemma":[0.000019141231,0.000049818562,0.0011290484,0.00038654442,0.000016910466,0.00010916975,0.00005855016,0.000150841,0.00011110434,0.002231985,0.9957203,0.000016526254],"about_ca_topic_score_codex":0.0020874636,"about_ca_topic_score_gemma":0.0034146672,"teacher_disagreement_score":0.12590815,"about_ca_system_score_codex":0.0017893116,"about_ca_system_score_gemma":0.0023403016,"threshold_uncertainty_score":0.42120475},"labels":[],"label_agreement":null},{"id":"W2034628356","doi":"10.1007/s10664-005-1290-x","title":"Studying Software Engineers: Data Collection Techniques for Software Field Studies","year":2005,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":491,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; University of Ottawa","funders":"","keywords":"Computer science; Software engineering; Field (mathematics); Software; Task (project management); Data science; Taxonomy (biology); Data collection; Systems engineering; Engineering","score_opus":0.09495362100580118,"score_gpt":0.36248098551184676,"score_spread":0.2675273645060456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034628356","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.141126,0.0014987752,0.6978443,0.0014994371,0.00039385862,0.08746344,0.046235953,0.0030672967,0.020870833],"genre_scores_gemma":[0.13007933,0.00124823,0.6530349,0.0007559146,0.00026201716,0.18638216,0.023060998,0.0006456068,0.004530814],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9390138,0.030056076,0.0122241,0.0044859825,0.012546178,0.0016739687],"domain_scores_gemma":[0.6642607,0.19355331,0.019353226,0.049397107,0.069742106,0.0036935757],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.045168616,0.0020172775,0.0024438668,0.023050098,0.0033682752,0.0031433005,0.003178183,0.0018309392,0.0064847814],"category_scores_gemma":[0.19732668,0.001381583,0.0018618354,0.022674654,0.002129896,0.0042026807,0.0042202366,0.003355481,0.0038778344],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011508809,0.00425472,0.13415684,0.008167575,0.0003675188,0.00046728717,0.02112395,0.003741004,0.01989563,0.017866176,0.062407594,0.7264008],"study_design_scores_gemma":[0.0020502529,0.0043956554,0.42725572,0.0042081946,0.0015523143,0.0010776045,0.045735262,0.030360175,0.08405102,0.05901596,0.33938572,0.0009121765],"about_ca_topic_score_codex":0.0043154215,"about_ca_topic_score_gemma":0.0068305763,"teacher_disagreement_score":0.95483136,"about_ca_system_score_codex":0.0026644887,"about_ca_system_score_gemma":0.0076513425,"threshold_uncertainty_score":0.23887736},"labels":[],"label_agreement":null},{"id":"W2039052699","doi":"10.1007/s10664-012-9228-6","title":"Studying re-opened bugs in open source software","year":2012,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":138,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Queen's University","funders":"","keywords":"Software bug; Eclipse; Software regression; Dimension (graph theory); Open source; Rework; Computer science; Software; Software quality; Software engineering; Engineering; Software development; Operating system; Mathematics","score_opus":0.05923128895567467,"score_gpt":0.32261791254277666,"score_spread":0.26338662358710196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2039052699","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99785393,0.0001623411,0.0011506287,0.00010115244,0.0000041818416,0.000008830975,0.000027357179,0.000018001023,0.0006734681],"genre_scores_gemma":[0.99822384,0.00010575931,0.0009825827,0.000022579601,0.0000065005506,0.000008775873,0.000111088135,0.000025320871,0.000513525],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9965423,0.0013531797,0.00028895805,0.000429693,0.0010822909,0.00030362935],"domain_scores_gemma":[0.8241104,0.12869783,0.024886852,0.008768903,0.011185049,0.0023510705],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046089008,0.00035011285,0.00032809438,0.002582762,0.000849801,0.001363404,0.0011074973,0.0011209928,0.0020648893],"category_scores_gemma":[0.10787618,0.00042104317,0.00040849167,0.0021825014,0.0011373146,0.0041203625,0.0013390464,0.0018862034,0.00023852078],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003537371,0.0020977797,0.8862586,0.00030938679,0.00023673216,0.0007326549,0.01151197,0.0037664045,0.004047874,0.005303929,0.0010836805,0.084297225],"study_design_scores_gemma":[0.00006626383,0.0011094655,0.9323799,0.00026409072,0.00019900955,0.0011947152,0.017681899,0.026572244,0.004678101,0.012496293,0.0032880763,0.000069960064],"about_ca_topic_score_codex":0.006470883,"about_ca_topic_score_gemma":0.011809086,"teacher_disagreement_score":0.006470883,"about_ca_system_score_codex":0.0010018197,"about_ca_system_score_gemma":0.0011527039,"threshold_uncertainty_score":0.024374485},"labels":[],"label_agreement":null},{"id":"W2040411904","doi":"10.1007/s10664-014-9312-1","title":"Do topics make sense to managers and developers?","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Traceability; Latent Dirichlet allocation; Computer science; Documentation; Relevance (law); Requirements traceability; Perception; Topic model; Control (management); Software engineering; Requirements engineering; Requirements elicitation; BitTorrent tracker; Software; Data science; Information retrieval; Requirement; Artificial intelligence","score_opus":0.019201760776928176,"score_gpt":0.26889327776064786,"score_spread":0.2496915169837197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2040411904","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.79928917,0.013769214,0.012090504,0.07471633,0.0034269597,0.00013910857,0.0010378245,0.00027192663,0.09525901],"genre_scores_gemma":[0.98537934,0.0035438214,0.002184164,0.0034348485,0.0009290126,0.00007738402,0.00036634918,0.00022233705,0.003862812],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9921967,0.003692909,0.00042102148,0.001020517,0.0015989419,0.0010699312],"domain_scores_gemma":[0.93719476,0.033595998,0.010698068,0.003962969,0.0064100586,0.008138141],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008111592,0.0004849129,0.00061054726,0.0040911636,0.0026077083,0.012790399,0.0011270333,0.0025193563,0.0090684695],"category_scores_gemma":[0.07862103,0.0005182946,0.0005448722,0.005896791,0.004745812,0.022037033,0.004561534,0.002389319,0.0019104952],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007149133,0.00040128,0.409554,0.0013834756,0.000252793,0.0008177864,0.20411879,0.00028319898,0.0037858805,0.10394907,0.042358287,0.23238054],"study_design_scores_gemma":[0.00019297842,0.00038679512,0.3318449,0.0017625387,0.00041899094,0.00084679225,0.3458059,0.0010553626,0.002018507,0.16930121,0.14624043,0.00012564335],"about_ca_topic_score_codex":0.0024894467,"about_ca_topic_score_gemma":0.0029314095,"teacher_disagreement_score":0.012790399,"about_ca_system_score_codex":0.0022367279,"about_ca_system_score_gemma":0.003408029,"threshold_uncertainty_score":0.042898715},"labels":[],"label_agreement":null},{"id":"W2043780253","doi":"10.1007/s10664-010-9131-y","title":"An empirical investigation into open source web applications’ implementation vulnerabilities","year":2010,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Web Application Security Vulnerabilities","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Open source; Computer science; World Wide Web; Web application; Empirical research; Computer security; Operating system; Software","score_opus":0.022168539556753533,"score_gpt":0.33476904965194676,"score_spread":0.31260051009519324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2043780253","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9976507,0.000039039503,0.00047997755,0.0001306461,0.0000014289114,0.000027886761,0.000034036573,0.0000064836654,0.0016298427],"genre_scores_gemma":[0.9988996,0.000053908167,0.0005797139,0.000037537688,0.000002701461,0.000026394418,0.000050628034,0.000005773866,0.0003436934],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.996326,0.0013794772,0.00036499804,0.00029983948,0.0012346011,0.00039504972],"domain_scores_gemma":[0.82321537,0.13037688,0.026381956,0.0061362963,0.012043274,0.0018461298],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0058634286,0.00025622864,0.00015187553,0.0035022518,0.0011325824,0.0012669361,0.0007109074,0.0009467788,0.0025882425],"category_scores_gemma":[0.07508699,0.00042964544,0.00022774411,0.0026762036,0.0019480397,0.004475364,0.0016545068,0.0017903075,0.00033278283],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019880375,0.0021552516,0.93538046,0.00013950275,0.00004792267,0.00075234467,0.019997437,0.000436193,0.0021002358,0.0067676413,0.0006900933,0.031334143],"study_design_scores_gemma":[0.000037537346,0.00064583216,0.9481562,0.00016334522,0.00006777658,0.0015094377,0.034553975,0.004655705,0.0030321812,0.0034385128,0.0037053511,0.000034239605],"about_ca_topic_score_codex":0.0024082966,"about_ca_topic_score_gemma":0.0049986523,"teacher_disagreement_score":0.0058634286,"about_ca_system_score_codex":0.0010611932,"about_ca_system_score_gemma":0.0018397121,"threshold_uncertainty_score":0.031009138},"labels":[],"label_agreement":null},{"id":"W2044807274","doi":"10.1007/s10664-010-9151-7","title":"Design evolution metrics for defect prediction in object oriented systems","year":2010,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Software metric; Eclipse; Data mining; Software quality; Quality (philosophy); Identification (biology); Software; Software system; Reliability engineering; Machine learning; Software engineering; Software development; Programming language; Engineering","score_opus":0.026845936819820337,"score_gpt":0.27604758517063077,"score_spread":0.24920164835081043,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2044807274","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7136414,0.0017537595,0.28008926,0.00034148642,0.00006241484,0.00012693563,0.0006275161,0.0010898727,0.0022672801],"genre_scores_gemma":[0.9576934,0.00013739309,0.04125258,0.000017616814,0.000013890363,0.000052375162,0.0005364063,0.00004906827,0.00024729627],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9949469,0.0020686311,0.00049699633,0.0003032809,0.002017994,0.00016630343],"domain_scores_gemma":[0.94753146,0.03561074,0.006036671,0.0036512038,0.0063079596,0.0008619447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006024601,0.000837305,0.0007142458,0.0065585733,0.00035795808,0.0009885108,0.0008169054,0.000871418,0.0007256448],"category_scores_gemma":[0.051922303,0.00031655942,0.00046581565,0.0026513012,0.00049301086,0.0021722037,0.0008167688,0.0008065454,0.00013949182],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044989784,0.0005311737,0.26521513,0.00033824253,0.00033295923,0.00008933922,0.0003206797,0.23625277,0.006020706,0.0064942944,0.0023338534,0.48162097],"study_design_scores_gemma":[0.00003282504,0.0006095235,0.042155113,0.000052727904,0.00009597361,0.00011720542,0.000070511654,0.94567645,0.0036427355,0.0069818,0.00053841044,0.00002670314],"about_ca_topic_score_codex":0.0026465973,"about_ca_topic_score_gemma":0.002880551,"teacher_disagreement_score":0.0065585733,"about_ca_system_score_codex":0.0010087567,"about_ca_system_score_gemma":0.0007902145,"threshold_uncertainty_score":0.031861484},"labels":[],"label_agreement":null},{"id":"W2045837563","doi":"10.1007/s10664-012-9219-7","title":"Static test case prioritization using topic models","year":2012,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":156,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Test suite; Computer science; Code coverage; Test case; Source code; Black box; Test Management Approach; Programming language; White-box testing; Test (biology); Test data; Operating system; Software; Software development; Artificial intelligence; Machine learning","score_opus":0.06432137932201937,"score_gpt":0.3036697114581389,"score_spread":0.23934833213611956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2045837563","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15312707,0.00089590263,0.82645845,0.00080564176,0.00013584176,0.00088188215,0.0015836222,0.009680969,0.0064306594],"genre_scores_gemma":[0.78265876,0.00029060998,0.20888586,0.00012865239,0.00013079308,0.00074050645,0.0029775435,0.0011024416,0.0030848272],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98874354,0.0058841496,0.0008405673,0.001650548,0.002206416,0.0006749231],"domain_scores_gemma":[0.9297371,0.056032773,0.0021194858,0.003859625,0.007100894,0.0011501584],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010082978,0.0017337598,0.0017287617,0.011113924,0.0010781361,0.003485714,0.0024311128,0.0019223097,0.007127939],"category_scores_gemma":[0.06890455,0.0010141858,0.0025149765,0.0052286875,0.00052674615,0.0047024703,0.0020739667,0.00204661,0.001583351],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022984557,0.0011208709,0.07243875,0.0011900079,0.00088609726,0.00049733196,0.0028960344,0.10889537,0.012467312,0.022688368,0.017262397,0.75735897],"study_design_scores_gemma":[0.00022694818,0.000357278,0.010594473,0.00009439843,0.00060665875,0.00033768095,0.0006551347,0.9425092,0.007078066,0.03263905,0.004808185,0.00009302979],"about_ca_topic_score_codex":0.008395957,"about_ca_topic_score_gemma":0.0128165735,"teacher_disagreement_score":0.011113924,"about_ca_system_score_codex":0.0019911523,"about_ca_system_score_gemma":0.0034481368,"threshold_uncertainty_score":0.05332452},"labels":[],"label_agreement":null},{"id":"W2046022618","doi":"10.1023/b:emse.0000013514.19567.ad","title":"An Industrial Case Study of Immediate Benefits of Requirements Engineering Process Improvement at the Australian Center for Unisys Software","year":2004,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Requirements management; Process management; Engineering; Process (computing); Requirements engineering; Requirements elicitation; Capability Maturity Model; Session (web analytics); Dimension (graph theory); Systems engineering; Engineering management; Computer science; Software","score_opus":0.06675324205872442,"score_gpt":0.3308906027842591,"score_spread":0.2641373607255347,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2046022618","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9798803,0.0001247024,0.008504152,0.0006026429,0.000017776369,0.00039393044,0.000061690735,0.00011646845,0.010298359],"genre_scores_gemma":[0.9845701,0.00009751877,0.0111882575,0.00008783778,0.000007713542,0.000087882356,0.00006094432,0.000028062996,0.003871663],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99516475,0.0028820038,0.00013054397,0.00027335138,0.0010327464,0.0005166488],"domain_scores_gemma":[0.9818943,0.012019983,0.00095773867,0.0016388297,0.0022011576,0.001288019],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0055225594,0.00043549208,0.00032579812,0.0010603706,0.0036674354,0.0014004497,0.0019910394,0.0022145852,0.0026826907],"category_scores_gemma":[0.014376814,0.00041927263,0.00045054205,0.0014695374,0.0012696559,0.0012755382,0.0015401205,0.0017748618,0.00044126454],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0049953344,0.03526417,0.08706668,0.0020742093,0.00027706256,0.046742324,0.13136676,0.08331602,0.058593582,0.027160557,0.015360741,0.50778246],"study_design_scores_gemma":[0.0021737665,0.03496102,0.29623172,0.00082770095,0.0005794525,0.013245502,0.16019237,0.24402478,0.09642504,0.014208974,0.13644876,0.0006809775],"about_ca_topic_score_codex":0.01639751,"about_ca_topic_score_gemma":0.032695603,"teacher_disagreement_score":0.99447745,"about_ca_system_score_codex":0.002755837,"about_ca_system_score_gemma":0.0026633788,"threshold_uncertainty_score":0.0326041},"labels":[],"label_agreement":null},{"id":"W2046157646","doi":"10.1007/s10664-015-9377-5","title":"An empirical study of software release notes","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Empirical research; Computer science; Software; Software engineering; Programming language; Mathematics; Statistics","score_opus":0.05553808588465858,"score_gpt":0.33739025458356287,"score_spread":0.2818521686989043,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2046157646","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9962166,0.00014002995,0.00039026293,0.00014098483,0.0000052673813,0.000029145434,0.00015134005,0.000012166576,0.002914267],"genre_scores_gemma":[0.997127,0.00014870186,0.00041150747,0.00005258419,0.000010960087,0.000020457595,0.00036821322,0.000011430103,0.0018491136],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9963671,0.0014713402,0.00030052432,0.00032417045,0.0012669765,0.00026991492],"domain_scores_gemma":[0.8280749,0.12994438,0.02068837,0.005924765,0.011597888,0.003769697],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.004191337,0.00023331062,0.00019826245,0.0023960355,0.0011948597,0.0017062909,0.0010298648,0.0006020519,0.005437764],"category_scores_gemma":[0.06375992,0.00027278898,0.00017046431,0.00268573,0.0011237711,0.0028568585,0.001039571,0.0014197918,0.0012073813],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005237376,0.003656942,0.9043496,0.00034440312,0.000050926206,0.00087235303,0.020885907,0.00037578196,0.0022400983,0.0043486585,0.0031625356,0.059189096],"study_design_scores_gemma":[0.000040663792,0.00090906327,0.95967484,0.00015410886,0.000043060336,0.00060586876,0.027436469,0.0016948346,0.0017003284,0.00069371844,0.007013414,0.000033562654],"about_ca_topic_score_codex":0.0047257757,"about_ca_topic_score_gemma":0.0074738995,"teacher_disagreement_score":0.99580866,"about_ca_system_score_codex":0.0009963281,"about_ca_system_score_gemma":0.0011620091,"threshold_uncertainty_score":0.022166193},"labels":[],"label_agreement":null},{"id":"W2049875335","doi":"10.1007/s10664-011-9179-3","title":"Computer-mediated communication to support distributed requirements elicitations and negotiations tasks","year":2011,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Requirements elicitation; Computer science; Negotiation; Technical communication; Knowledge management; Empirical research; Requirements management; Stakeholder; Requirements engineering; Process management; Software; Engineering","score_opus":0.06055152737650241,"score_gpt":0.30205757710763703,"score_spread":0.24150604973113463,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2049875335","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44771114,0.00026037818,0.49728334,0.0012455294,0.00014440356,0.000948092,0.0002610729,0.0050743865,0.04707162],"genre_scores_gemma":[0.88108605,0.00009232168,0.10919056,0.00019878465,0.00006441578,0.0008061496,0.00023835464,0.000214435,0.008109052],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9919836,0.006162923,0.00022333766,0.00041504606,0.0009839197,0.00023114025],"domain_scores_gemma":[0.92475706,0.063724615,0.0019103548,0.0058402224,0.0029666976,0.000801095],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006296092,0.0006144125,0.00036277538,0.0011649161,0.0012987637,0.0025028882,0.0017083698,0.002048978,0.011921777],"category_scores_gemma":[0.04117043,0.00042323652,0.00025904016,0.0007574113,0.0005510886,0.002488308,0.0027109976,0.0011529273,0.002319643],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002831556,0.003940918,0.01009351,0.0009566942,0.00023509789,0.0019939952,0.023714015,0.03335119,0.078715414,0.043885652,0.017916678,0.78236526],"study_design_scores_gemma":[0.0018121033,0.0050375443,0.022431219,0.00094452745,0.00058037677,0.0025820627,0.013371844,0.6452945,0.11697915,0.07179667,0.11867588,0.0004940451],"about_ca_topic_score_codex":0.0010551614,"about_ca_topic_score_gemma":0.0015128937,"teacher_disagreement_score":0.011921777,"about_ca_system_score_codex":0.00090865855,"about_ca_system_score_gemma":0.0013979798,"threshold_uncertainty_score":0.039882243},"labels":[],"label_agreement":null},{"id":"W2051162588","doi":"10.1007/s10664-014-9311-2","title":"Modelling the ‘hurried’ bug report reading process to summarize bug reports","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":70,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Automatic summarization; Computer science; Process (computing); Sentence; Reading (process); Software bug; Quality (philosophy); Natural language processing; Artificial intelligence; Data science; Information retrieval; Software; Programming language; Linguistics","score_opus":0.02334213703925641,"score_gpt":0.2861944876746135,"score_spread":0.26285235063535706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2051162588","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44097272,0.0007962775,0.54322314,0.0020539125,0.00016755122,0.00028242107,0.001369849,0.0028640945,0.008269981],"genre_scores_gemma":[0.93161625,0.00019946668,0.06218414,0.00008709277,0.000049775208,0.00010001551,0.00077458954,0.00018188288,0.004806778],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981724,0.00080582715,0.00014335568,0.0004371361,0.0002691523,0.00017205552],"domain_scores_gemma":[0.9665,0.025112558,0.003486669,0.0019020297,0.0023627107,0.0006360778],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042259595,0.0008133941,0.00071092503,0.0020083636,0.00040728628,0.0034535336,0.0014670703,0.0025634237,0.0060080932],"category_scores_gemma":[0.03912948,0.00080244103,0.00097105687,0.0016603683,0.00088999985,0.0028389222,0.001029312,0.0017833164,0.0014329753],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009620198,0.00047568124,0.039517857,0.00042191579,0.00021441848,0.0006760194,0.00263123,0.79119,0.0064581838,0.04873893,0.0042136293,0.10450011],"study_design_scores_gemma":[0.00002636349,0.00008667171,0.0032053678,0.000019217161,0.000038641072,0.000058612524,0.00008770356,0.9849858,0.0007634799,0.009844532,0.0008576593,0.00002601364],"about_ca_topic_score_codex":0.020050233,"about_ca_topic_score_gemma":0.014845828,"teacher_disagreement_score":0.020050233,"about_ca_system_score_codex":0.0012605392,"about_ca_system_score_gemma":0.0017350394,"threshold_uncertainty_score":0.039867043},"labels":[],"label_agreement":null},{"id":"W2052776473","doi":"10.1007/s10664-014-9324-x","title":"A Large-Scale Empirical Study of the Relationship between Build Technology and Build Maintenance","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Queen's University","funders":"","keywords":"Software maintenance; Computer science; Deliverable; Abstraction; Software engineering; Software; Source code; Scale (ratio); Source lines of code; Code (set theory); Systems engineering; Software system; Engineering; Programming language","score_opus":0.026836341569620767,"score_gpt":0.2976999388033339,"score_spread":0.2708635972337131,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2052776473","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9969703,0.00021318528,0.00081529916,0.00013948762,0.0000048142574,0.000021180169,0.00015128753,0.000017463153,0.0016667956],"genre_scores_gemma":[0.99848807,0.00014433922,0.0006451793,0.000055731263,0.000009492316,0.00001710536,0.00024485908,0.0000067876836,0.00038839778],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9986411,0.0005935642,0.00008202243,0.00017356433,0.00041374261,0.000096060154],"domain_scores_gemma":[0.9019074,0.07541063,0.011575528,0.0043220217,0.0045682206,0.0022161158],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002700774,0.00028766092,0.00024673907,0.0017194581,0.0008210525,0.0008258102,0.00065107807,0.0006782079,0.0028939424],"category_scores_gemma":[0.03367406,0.0003323713,0.00031780012,0.0022451538,0.00087354414,0.0013683151,0.00071592565,0.0012223044,0.000581597],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002459554,0.0023880396,0.9618664,0.00016655083,0.00024354868,0.00026019372,0.0017715917,0.0010079507,0.00095154863,0.0013941819,0.001428592,0.028275458],"study_design_scores_gemma":[0.000026393047,0.00041535654,0.99358714,0.000043480217,0.00009270181,0.0001787447,0.0015779784,0.0020669152,0.0004749079,0.00035198047,0.0011709661,0.000013450276],"about_ca_topic_score_codex":0.006204452,"about_ca_topic_score_gemma":0.013121573,"teacher_disagreement_score":0.006204452,"about_ca_system_score_codex":0.00069093774,"about_ca_system_score_gemma":0.0011041552,"threshold_uncertainty_score":0.01428324},"labels":[],"label_agreement":null},{"id":"W2055777130","doi":"10.1007/s10664-015-9366-8","title":"Investigating technical and non-technical factors influencing modern code review","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Computer science; Code review; Process (computing); Code (set theory); Key (lock); Source code; Variety (cybernetics); Empirical research; Software engineering; Component (thermodynamics); Data science; Static program analysis; Software development; Computer security; Software; Artificial intelligence; Programming language","score_opus":0.05327895845453611,"score_gpt":0.31370679174594,"score_spread":0.2604278332914039,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2055777130","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98552126,0.0024286374,0.001735596,0.0019489169,0.00005005244,0.000085387896,0.00016868455,0.000049709888,0.008011801],"genre_scores_gemma":[0.9971733,0.0006378111,0.0008177828,0.00017349693,0.000044531145,0.000022472981,0.00010556763,0.000032555457,0.0009923936],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.97902244,0.008113833,0.0022548165,0.0014885828,0.0076214746,0.0014988586],"domain_scores_gemma":[0.3626739,0.42215273,0.12708075,0.0099031385,0.069559604,0.008629865],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.022815673,0.00021242262,0.0003273773,0.0069289524,0.0014174397,0.004311479,0.0009769073,0.0008618725,0.0039581154],"category_scores_gemma":[0.35692203,0.00033275777,0.0004785622,0.006150205,0.0016077352,0.0036506841,0.0015031842,0.0015136322,0.0005186999],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003051897,0.00017139602,0.9261365,0.00047165676,0.00020282765,0.0002795087,0.005334309,0.0005329262,0.0012861066,0.002957347,0.001879202,0.060443003],"study_design_scores_gemma":[0.000015424559,0.00018728268,0.98319995,0.00026167452,0.00013048579,0.0003667104,0.005194132,0.0015114064,0.0012012728,0.0014032105,0.006487501,0.00004097562],"about_ca_topic_score_codex":0.010609989,"about_ca_topic_score_gemma":0.02444933,"teacher_disagreement_score":0.97718436,"about_ca_system_score_codex":0.0033962973,"about_ca_system_score_gemma":0.007930938,"threshold_uncertainty_score":0.12066227},"labels":[],"label_agreement":null},{"id":"W2056894403","doi":"10.1007/s10664-012-9231-y","title":"What are developers talking about? An analysis of topics and trends in Stack Overflow","year":2012,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":613,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Latent Dirichlet allocation; World Wide Web; Topic model; Popularity; Data science; Leverage (statistics); Android (operating system); Information retrieval; Artificial intelligence","score_opus":0.03051975966812,"score_gpt":0.3086728286717767,"score_spread":0.2781530690036567,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056894403","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.995082,0.00091489876,0.000580288,0.0008720514,0.00001657954,0.000017983984,0.00045520058,0.00003546379,0.0020255374],"genre_scores_gemma":[0.99575776,0.0012105553,0.0010224718,0.0001719724,0.00007407573,0.000033321685,0.00085620594,0.000052204992,0.00082148396],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99718946,0.0008733026,0.00034957347,0.0003361222,0.000933444,0.00031806846],"domain_scores_gemma":[0.9128747,0.056387976,0.01573795,0.0011747868,0.010728298,0.003096273],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0037436222,0.00022085135,0.00025373825,0.0086217085,0.0011083912,0.0021346668,0.00051845395,0.0007278773,0.0011728773],"category_scores_gemma":[0.041532904,0.0003245859,0.00029204867,0.008282863,0.0008052008,0.0043893224,0.001333264,0.0010974403,0.00027353564],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032378393,0.00013002686,0.8414324,0.00040087625,0.000055258413,0.00040967812,0.065325655,0.00022459858,0.0032429835,0.0014555064,0.0026769992,0.084322244],"study_design_scores_gemma":[0.000011042254,0.00010615956,0.9496954,0.00021500561,0.00007424177,0.00046980561,0.039001185,0.0013763714,0.0011681066,0.00070551404,0.0071460414,0.00003119017],"about_ca_topic_score_codex":0.009625689,"about_ca_topic_score_gemma":0.013259637,"teacher_disagreement_score":0.99625635,"about_ca_system_score_codex":0.0016067321,"about_ca_system_score_gemma":0.0024095215,"threshold_uncertainty_score":0.019798398},"labels":[],"label_agreement":null},{"id":"W2061072593","doi":"10.1007/s10664-013-9264-x","title":"SWordNet: Inferring semantically related words from software context","year":2013,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":59,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; WordNet; Program comprehension; Information retrieval; Software; Context (archaeology); Java; Ranking (information retrieval); Natural language processing; Code (set theory); Precision and recall; Software maintenance; Artificial intelligence; Software development; Programming language; Software system","score_opus":0.01656489003974,"score_gpt":0.2521907926293967,"score_spread":0.2356259025896567,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2061072593","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21641308,0.00352391,0.6703291,0.0012360181,0.0009333715,0.00080794404,0.02845489,0.06250844,0.015793275],"genre_scores_gemma":[0.4404377,0.0013915823,0.4990925,0.0005061411,0.0002479478,0.00059174455,0.05028054,0.00279609,0.004655763],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986125,0.00037252996,0.00015791129,0.0005079021,0.00025218853,0.000097077515],"domain_scores_gemma":[0.9970419,0.0018110266,0.00019191412,0.00049864687,0.00033806675,0.00011847414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009817193,0.002267074,0.0011620562,0.0052760183,0.0014086638,0.0018138845,0.0015482788,0.001924787,0.008686532],"category_scores_gemma":[0.0063138558,0.0010282712,0.0015732356,0.0033536877,0.00092571386,0.008540773,0.0043499176,0.001687803,0.0039266692],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022355781,0.00085889647,0.022455452,0.0035993063,0.00076027703,0.0026863473,0.0033893643,0.015479437,0.051400114,0.05451141,0.08209397,0.76052994],"study_design_scores_gemma":[0.00048088137,0.0005886122,0.012556946,0.00061366806,0.0008967371,0.0019379924,0.004212559,0.5445394,0.039241817,0.28879505,0.105893634,0.00024266631],"about_ca_topic_score_codex":0.007772508,"about_ca_topic_score_gemma":0.014463442,"teacher_disagreement_score":0.008686532,"about_ca_system_score_codex":0.00078143796,"about_ca_system_score_gemma":0.0019265283,"threshold_uncertainty_score":0.02905935},"labels":[],"label_agreement":null},{"id":"W2062937278","doi":"10.1023/b:emse.0000048324.12188.a2","title":"An Empirical Exploration of the Distributions of the Chidamber and Kemerer Object-Oriented Metrics Suite","year":2004,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":71,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Collinearity; Suite; Computer science; Data mining; Correlation; Parametric statistics; Variance (accounting); Empirical research; Set (abstract data type); Test suite; Regression analysis; Statistics; Machine learning; Mathematics; Test case","score_opus":0.0281820532597251,"score_gpt":0.2982924489141738,"score_spread":0.27011039565444867,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2062937278","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9164433,0.00083114253,0.07631865,0.0010597183,0.000016635322,0.000066115266,0.0005849218,0.00019548622,0.0044841375],"genre_scores_gemma":[0.98582876,0.0002103454,0.012537356,0.00005689905,0.00002014994,0.000047191174,0.00075115665,0.00008350517,0.00046473255],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98149085,0.011507259,0.0007536313,0.0015488712,0.004302619,0.0003967141],"domain_scores_gemma":[0.5578032,0.3902953,0.011434882,0.022157105,0.016227,0.002082508],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.028988527,0.0005234363,0.0005536204,0.004895219,0.00070304354,0.0024080777,0.0019548424,0.0011193297,0.0022200085],"category_scores_gemma":[0.3402905,0.00040797252,0.00045686623,0.0049625062,0.002396953,0.0056342897,0.0020290646,0.002273678,0.00049777306],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000724419,0.00062859553,0.5852708,0.00025423968,0.00024809717,0.0004117664,0.006678092,0.046998706,0.0029213836,0.16160482,0.006714009,0.18754509],"study_design_scores_gemma":[0.00014483366,0.00074655045,0.36486873,0.0002013351,0.00011520736,0.0023522975,0.004154884,0.42747724,0.003851355,0.18393463,0.011999036,0.0001538804],"about_ca_topic_score_codex":0.0022539352,"about_ca_topic_score_gemma":0.002476842,"teacher_disagreement_score":0.97101146,"about_ca_system_score_codex":0.0015630599,"about_ca_system_score_gemma":0.0011017675,"threshold_uncertainty_score":0.1533078},"labels":[],"label_agreement":null},{"id":"W2066687630","doi":"10.1007/s10664-013-9292-6","title":"Towards improving statistical modeling of software engineering data: think locally, act globally!","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Cluster analysis; Multivariate adaptive regression splines; Data mining; Machine learning; Software; Statistical model; Parametric statistics; Data modeling; Artificial intelligence; Data science; Regression analysis; Nonparametric regression; Mathematics; Statistics; Software engineering","score_opus":0.03219726578772242,"score_gpt":0.2902385702837676,"score_spread":0.2580413044960452,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2066687630","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0042074416,0.0005568468,0.97375315,0.018446777,0.00016746964,0.00005312968,0.0001172218,0.0014743087,0.001223631],"genre_scores_gemma":[0.08202575,0.0010688298,0.90793806,0.005238143,0.0004739096,0.00025032272,0.00040459188,0.001254008,0.0013463143],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.94334006,0.04207999,0.0018262382,0.0041256105,0.007812378,0.0008157023],"domain_scores_gemma":[0.7297781,0.1462482,0.012823354,0.081868425,0.025311468,0.003970454],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07680857,0.00283248,0.0036348812,0.0043869806,0.0018545771,0.011362925,0.004437484,0.0054854727,0.0047256076],"category_scores_gemma":[0.24037509,0.002251046,0.0032083932,0.0046262937,0.007781678,0.034479827,0.008205913,0.015933225,0.0041406853],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029825518,0.0008558477,0.041392874,0.0012372265,0.0015482021,0.00018403976,0.00407631,0.07187728,0.008668546,0.24796177,0.043667864,0.5782317],"study_design_scores_gemma":[0.00007392813,0.00017875353,0.0036745905,0.00053571776,0.00021896724,0.00010608142,0.0016713551,0.24423127,0.0051847664,0.7248739,0.01909667,0.00015395989],"about_ca_topic_score_codex":0.007983583,"about_ca_topic_score_gemma":0.008212539,"teacher_disagreement_score":0.07680857,"about_ca_system_score_codex":0.002548395,"about_ca_system_score_gemma":0.008333705,"threshold_uncertainty_score":0.40620738},"labels":[],"label_agreement":null},{"id":"W2068712694","doi":"10.1007/s10664-009-9121-0","title":"Applying empirical software engineering to software architecture: challenges and lessons learned","year":2009,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Social software engineering; Resource-oriented architecture; Software development; Software peer review; Personal software process; Software construction; Architecture tradeoff analysis method; Software architecture; Reference architecture; Software Engineering Process Group","score_opus":0.10052347010825755,"score_gpt":0.34551098358241233,"score_spread":0.2449875134741548,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2068712694","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0818376,0.07245022,0.45231903,0.34940654,0.0021648344,0.00023856887,0.00011091442,0.00030212267,0.04117021],"genre_scores_gemma":[0.69142747,0.09089811,0.19578163,0.012140142,0.0031655096,0.00036323688,0.00017332357,0.00019737153,0.005853251],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.9811122,0.013948144,0.00070183247,0.00076329004,0.003023296,0.00045131365],"domain_scores_gemma":[0.8015445,0.168737,0.0023058173,0.011436803,0.013861444,0.0021143511],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.036498293,0.00092835346,0.0012173807,0.0023774633,0.0014913866,0.0073236316,0.0035621112,0.0031657426,0.0028040025],"category_scores_gemma":[0.10860463,0.0007337246,0.0006138164,0.003130285,0.0121655455,0.015231519,0.0041209203,0.008341824,0.0006133453],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004273553,0.00065405626,0.011510222,0.0011373254,0.000106773994,0.00019342314,0.002776111,0.009979209,0.00050338364,0.6147612,0.008886054,0.34944955],"study_design_scores_gemma":[0.000025063173,0.00006957977,0.002148788,0.0010543867,0.000011863644,0.00010874857,0.0047686924,0.012680189,0.00043244386,0.95518047,0.023494925,0.000024788476],"about_ca_topic_score_codex":0.0046596816,"about_ca_topic_score_gemma":0.0056872144,"teacher_disagreement_score":0.036498293,"about_ca_system_score_codex":0.0029967811,"about_ca_system_score_gemma":0.007894983,"threshold_uncertainty_score":0.19302374},"labels":[],"label_agreement":null},{"id":"W2070873282","doi":"10.1007/s10664-011-9169-5","title":"The evolution of Java build systems","year":2011,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Java; Executable; Software evolution; Software maintenance; Software system; Software engineering; Source code; Software development; Legacy system; Overhead (engineering); Codebase; Source lines of code; Software; Operating system; Distributed computing; Software construction","score_opus":0.028469191643593714,"score_gpt":0.2551634501681812,"score_spread":0.2266942585245875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2070873282","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9871273,0.00026796685,0.0031248978,0.00075774675,0.000011773496,0.000015331463,0.00019734964,0.00006490428,0.008432845],"genre_scores_gemma":[0.99679923,0.00011045835,0.0015857483,0.000040421888,0.0000054261604,0.0000042842835,0.00017535474,0.000025642823,0.0012532683],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99857044,0.0005606273,0.00007959474,0.00018119194,0.00046843308,0.00013970204],"domain_scores_gemma":[0.9809176,0.010537713,0.0025044386,0.0019884747,0.0034304562,0.00062136014],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026556815,0.00015917062,0.00011203029,0.0014071802,0.00048589584,0.0018382899,0.00050928973,0.0006459558,0.0020588543],"category_scores_gemma":[0.036080834,0.0002888966,0.00019704274,0.0013632404,0.0008112526,0.0027978243,0.0007468712,0.0009841537,0.0002590135],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004377382,0.00053932786,0.66513884,0.0001742917,0.00013936554,0.00051617314,0.0048564975,0.03371859,0.008830003,0.046881385,0.0030188067,0.23574893],"study_design_scores_gemma":[0.000034920013,0.00022864016,0.8549366,0.00009024659,0.000095599156,0.00050848554,0.0027887698,0.10220103,0.0047083967,0.017986499,0.01637227,0.000048649574],"about_ca_topic_score_codex":0.012510412,"about_ca_topic_score_gemma":0.015489274,"teacher_disagreement_score":0.012510412,"about_ca_system_score_codex":0.0019933917,"about_ca_system_score_gemma":0.0009478204,"threshold_uncertainty_score":0.024875224},"labels":[],"label_agreement":null},{"id":"W2071255138","doi":"10.1007/s10664-014-9323-y","title":"Recommending reference API documentation","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":67,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Documentation; Computer science; Categorization; Programmer; Set (abstract data type); Internal documentation; Information retrieval; World Wide Web; Software; Artificial intelligence; Programming language; Software system","score_opus":0.03239301421113391,"score_gpt":0.30504308565920896,"score_spread":0.27265007144807507,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2071255138","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27092254,0.0065027867,0.5602111,0.0067295046,0.0027751254,0.0014860531,0.007873271,0.0626672,0.08083245],"genre_scores_gemma":[0.5664424,0.0028332015,0.36205223,0.0011645468,0.0006287945,0.00052105705,0.017681992,0.0036037862,0.04507205],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9946477,0.0015764543,0.00044806566,0.0006753393,0.0024434072,0.00020898004],"domain_scores_gemma":[0.96956563,0.007850987,0.0014458302,0.0057420204,0.014515078,0.0008804859],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003378434,0.0011279515,0.0010793498,0.010040276,0.001604274,0.002832006,0.0015979406,0.002183166,0.014816348],"category_scores_gemma":[0.06631322,0.000676202,0.0010086986,0.005870799,0.00034013236,0.004117116,0.0014610904,0.0017994692,0.00935573],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034934568,0.00060676876,0.016954489,0.00067614333,0.00010675158,0.00040169773,0.00048376885,0.005559518,0.00857547,0.004246269,0.14295493,0.8190849],"study_design_scores_gemma":[0.0004255502,0.001383113,0.037759885,0.0013473724,0.0008493579,0.002178494,0.002091787,0.5839574,0.048242595,0.027817177,0.29359892,0.00034850414],"about_ca_topic_score_codex":0.009660594,"about_ca_topic_score_gemma":0.01985606,"teacher_disagreement_score":0.014816348,"about_ca_system_score_codex":0.00097325904,"about_ca_system_score_gemma":0.0031696665,"threshold_uncertainty_score":0.049565613},"labels":[],"label_agreement":null},{"id":"W2072143380","doi":"10.1007/s10664-015-9361-0","title":"Evaluating the impact of design pattern and anti-pattern dependencies on changes and faults","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Polytechnique Montréal; Queen's University","funders":"Fonds Québécois de la Recherche sur la Nature et les Technologies; Canada Research Chairs","keywords":"Software design pattern; Structural pattern; Maintainability; Computer science; Design pattern; Flexibility (engineering); Architectural pattern; Data mining; Software design; Software; Software engineering; Software development; Programming language; Mathematics; Statistics","score_opus":0.1456268245412086,"score_gpt":0.381946844692815,"score_spread":0.23632002015160639,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2072143380","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99767333,0.00006247698,0.0015344806,0.000043630025,0.000006214417,0.000018390056,0.00015548793,0.000028558909,0.00047738315],"genre_scores_gemma":[0.99821955,0.000026899115,0.0013009548,0.000010177962,0.0000042850515,0.000011875188,0.00023599374,0.000008996217,0.00018128172],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9934384,0.002939383,0.0005505039,0.0010225208,0.0015974367,0.000451806],"domain_scores_gemma":[0.55813384,0.4070131,0.018788682,0.00803318,0.006345857,0.0016853496],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007817053,0.00057843403,0.00038214235,0.001579297,0.00029188304,0.0008285827,0.00097081694,0.0009020856,0.0022279676],"category_scores_gemma":[0.119632035,0.00034983057,0.00076677196,0.0011531146,0.00078420324,0.0021359092,0.00049693324,0.0013260002,0.00020548185],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0057651578,0.003569873,0.7863784,0.0002959987,0.00093956356,0.00029416688,0.00028427612,0.10370108,0.010143799,0.0014542062,0.0003659732,0.086807504],"study_design_scores_gemma":[0.00024730724,0.00663626,0.6837173,0.000047581758,0.00087779766,0.00033863456,0.0005827688,0.2877829,0.015787119,0.003251482,0.0006667935,0.00006415717],"about_ca_topic_score_codex":0.0029344524,"about_ca_topic_score_gemma":0.0055516623,"teacher_disagreement_score":0.007817053,"about_ca_system_score_codex":0.0009415587,"about_ca_system_score_gemma":0.0013098345,"threshold_uncertainty_score":0.041341066},"labels":[],"label_agreement":null},{"id":"W2074917598","doi":"10.1007/s10664-014-9353-5","title":"The effects of visualization and interaction techniques on feature model configuration","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University; Simon Fraser University","funders":"","keywords":"Feature model; Feature (linguistics); Software product line; Computer science; Visualization; Artifact (error); Process (computing); Software visualization; Domain (mathematical analysis); Feature-oriented domain analysis; Data mining; Software; Domain engineering; Domain analysis; Set (abstract data type); Software engineering; Artificial intelligence; Software system; Software development; Component-based software engineering; Software construction","score_opus":0.03807687604875666,"score_gpt":0.33070934099790766,"score_spread":0.292632464949151,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2074917598","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9866731,0.0009417957,0.008238924,0.00021513339,0.00008168175,0.000058534977,0.00017215169,0.000677266,0.0029414464],"genre_scores_gemma":[0.9925972,0.00019910575,0.005777658,0.000055901142,0.000042027037,0.00003623959,0.00012523847,0.0003147357,0.0008519091],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9905077,0.006511503,0.00077556877,0.00075466424,0.0009564592,0.00049410306],"domain_scores_gemma":[0.36348134,0.611238,0.0095401285,0.010100852,0.004140081,0.0014995937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008877357,0.0011349461,0.0007985571,0.0015384239,0.0006488802,0.0032939217,0.0012564121,0.0018699629,0.0072597945],"category_scores_gemma":[0.2411897,0.0008098358,0.00088543823,0.0016430591,0.0009099082,0.0059940987,0.0016389451,0.001819918,0.00060491054],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.067743376,0.008707765,0.1786585,0.002601331,0.0019021946,0.001509395,0.0054695006,0.13197187,0.14575215,0.0035843193,0.0037827203,0.4483169],"study_design_scores_gemma":[0.003325857,0.027758496,0.50457156,0.00082073576,0.0061029517,0.0033063476,0.0060044164,0.30653164,0.12020779,0.013749063,0.006761352,0.0008597899],"about_ca_topic_score_codex":0.0018533365,"about_ca_topic_score_gemma":0.001866496,"teacher_disagreement_score":0.008877357,"about_ca_system_score_codex":0.0005351694,"about_ca_system_score_gemma":0.0007382829,"threshold_uncertainty_score":0.046948493},"labels":[],"label_agreement":null},{"id":"W2075269190","doi":"10.1007/s10664-014-9350-8","title":"Linguistic antipatterns: what they are and how developers perceive them","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":116,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Lexicon; Documentation; Source code; Cognitive dissonance; Computer science; Open source; Code (set theory); Empirical research; Affect (linguistics); Code review; Data science; Linguistics; Psychology; Artificial intelligence; Software development; Static program analysis; Social psychology; Software; Programming language; Communication; Epistemology","score_opus":0.064961403368231,"score_gpt":0.28485054955810457,"score_spread":0.21988914618987357,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2075269190","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8677104,0.0006497867,0.08680447,0.0063673304,0.00014930079,0.00008770213,0.00024423256,0.0010462027,0.036940567],"genre_scores_gemma":[0.9760399,0.00027058745,0.019416343,0.00059609086,0.000054254375,0.00007086951,0.00018143734,0.00048590032,0.0028846038],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9929349,0.003056821,0.0005336074,0.00072047405,0.0024490906,0.00030507345],"domain_scores_gemma":[0.9493569,0.025655616,0.009115425,0.005201783,0.009469068,0.0012010881],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050787334,0.0003302138,0.00033196656,0.001464488,0.0009523791,0.0041667726,0.0007893291,0.0018530937,0.0032090703],"category_scores_gemma":[0.049852934,0.0006791536,0.00021097859,0.0010621698,0.0024716365,0.009136185,0.0020299493,0.0017453216,0.0007937949],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043220524,0.0003173319,0.34849787,0.00089623657,0.000104799845,0.0011336685,0.1929518,0.0007535458,0.064389646,0.089545704,0.010343952,0.29063323],"study_design_scores_gemma":[0.00014651663,0.0005513046,0.45618743,0.0010307007,0.00037786507,0.0036641317,0.19156946,0.022423,0.028994408,0.18630971,0.10848773,0.0002576612],"about_ca_topic_score_codex":0.0020137767,"about_ca_topic_score_gemma":0.0025442168,"teacher_disagreement_score":0.0050787334,"about_ca_system_score_codex":0.00067583774,"about_ca_system_score_gemma":0.0013805954,"threshold_uncertainty_score":0.026859224},"labels":[],"label_agreement":null},{"id":"W2087670325","doi":"10.1007/s10664-014-9333-9","title":"Improving bug management using correlations in crash reports","year":2014,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Queen's University","funders":"","keywords":"Crash; Computer science; Eclipse; Precision and recall; Identification (biology); Software bug; Software; Data mining; Artificial intelligence; Programming language","score_opus":0.019100434507084794,"score_gpt":0.2706619834027131,"score_spread":0.25156154889562826,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2087670325","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8861395,0.0017031514,0.10059976,0.00077263766,0.00020200286,0.00022309176,0.0020734675,0.0050206254,0.0032658041],"genre_scores_gemma":[0.9739226,0.00022983101,0.024063561,0.000048189577,0.00008437805,0.000054033317,0.0011055691,0.00012610995,0.00036574635],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99224687,0.0029354289,0.0007781533,0.001421268,0.0022321432,0.0003860571],"domain_scores_gemma":[0.863228,0.0720521,0.034564152,0.010034559,0.017743243,0.0023779462],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0063381977,0.0011649611,0.000981299,0.0095089935,0.0006593525,0.0019765613,0.0011451757,0.00095307844,0.0014938563],"category_scores_gemma":[0.105690345,0.0007408197,0.00076024345,0.0053568594,0.0004091624,0.0032567268,0.0011633864,0.0014288784,0.00078984944],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059626636,0.00075260305,0.7294303,0.00033673068,0.000478531,0.00020985573,0.00051808293,0.023633568,0.0056733014,0.001290148,0.006295149,0.23078535],"study_design_scores_gemma":[0.00015769099,0.0017469391,0.48210624,0.0002748009,0.000842472,0.0006027311,0.0007632187,0.49045667,0.012550717,0.0061450345,0.0041795317,0.00017400736],"about_ca_topic_score_codex":0.006217278,"about_ca_topic_score_gemma":0.010267488,"teacher_disagreement_score":0.0095089935,"about_ca_system_score_codex":0.0008428126,"about_ca_system_score_gemma":0.0022608333,"threshold_uncertainty_score":0.033519983},"labels":[],"label_agreement":null},{"id":"W2090432523","doi":"10.1007/s10664-008-9076-6","title":"“Cloning considered harmful” considered harmful: patterns of cloning in software","year":2008,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":363,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Cloning (programming); Source code; Computer science; Software engineering; Code (set theory); Software system; Software; clone (Java method); Maintainability; Web application; Codebase; Programming language; World Wide Web; Biology; Genetics","score_opus":0.044662327963924,"score_gpt":0.2815639292607685,"score_spread":0.2369016012968445,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2090432523","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9471351,0.00061161804,0.02840479,0.0043823253,0.000050760143,0.000067672154,0.00023786865,0.00016148841,0.01894835],"genre_scores_gemma":[0.9948925,0.000085587446,0.0040600584,0.00030649663,0.000016922082,0.000020732556,0.00008091426,0.000048955255,0.00048778555],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98219854,0.009560371,0.0017342173,0.0014244842,0.0036944733,0.0013879667],"domain_scores_gemma":[0.7696085,0.1448309,0.041810166,0.029270563,0.011050499,0.003429307],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012257816,0.00030273636,0.00054259703,0.003719789,0.0043408195,0.0040287087,0.0013750414,0.0023844843,0.003393452],"category_scores_gemma":[0.1295866,0.00049913855,0.00051375886,0.00789833,0.0094178235,0.0082446635,0.0051291357,0.0038294152,0.0003049181],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073648285,0.00022182506,0.56268966,0.00059916044,0.00024311933,0.0007302896,0.11433606,0.0015375875,0.0056407657,0.20881662,0.0054363683,0.099012084],"study_design_scores_gemma":[0.00008643669,0.00033463247,0.48848248,0.00089652865,0.00042711498,0.0044539226,0.10480571,0.008696132,0.0077897776,0.34992328,0.033862174,0.00024180974],"about_ca_topic_score_codex":0.0051151607,"about_ca_topic_score_gemma":0.0062507805,"teacher_disagreement_score":0.012257816,"about_ca_system_score_codex":0.0019879965,"about_ca_system_score_gemma":0.0034266147,"threshold_uncertainty_score":0.06482631},"labels":[],"label_agreement":null},{"id":"W2093400716","doi":"10.1007/s10664-015-9379-3","title":"What are mobile developers asking about? A large scale study using stack overflow","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":311,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Mobile device; World Wide Web; Mobile computing; Latent Dirichlet allocation; Context (archaeology); Software; Popularity; Mobile Web; Data science; Mobile technology; Software development; Topic model; Telecommunications; Artificial intelligence; Operating system","score_opus":0.0514213064798872,"score_gpt":0.32439770776851695,"score_spread":0.27297640128862977,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2093400716","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99888664,0.000049347036,0.00016807002,0.00018334737,0.000004059008,0.00003792654,0.00005055216,0.000007002263,0.0006130934],"genre_scores_gemma":[0.9979048,0.00016268711,0.0005234713,0.00029065105,0.0000131345005,0.00012050925,0.000117254014,0.000018863422,0.0008487525],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9968951,0.001451067,0.00024268884,0.00035509077,0.0006447904,0.00041123878],"domain_scores_gemma":[0.90798396,0.06840722,0.010575872,0.002007544,0.006842412,0.00418298],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057516918,0.0004887506,0.00043829912,0.0025253308,0.0031417918,0.0024425455,0.0008518639,0.0016634722,0.0018942838],"category_scores_gemma":[0.052636843,0.0007215972,0.0002750181,0.0021910467,0.0014851672,0.004181307,0.002157345,0.002351156,0.0005235501],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002817505,0.0021616563,0.7029215,0.00025769533,0.00006389122,0.0011832587,0.2563147,0.00011133351,0.0017727845,0.0007381536,0.0021138473,0.032079387],"study_design_scores_gemma":[0.00011568786,0.0009109539,0.6762496,0.00032482023,0.000116198695,0.0006376585,0.31242812,0.00087705394,0.0010933804,0.0005268846,0.006644589,0.00007508994],"about_ca_topic_score_codex":0.016766833,"about_ca_topic_score_gemma":0.040645234,"teacher_disagreement_score":0.016766833,"about_ca_system_score_codex":0.0018900449,"about_ca_system_score_gemma":0.0033947048,"threshold_uncertainty_score":0.033338487},"labels":[],"label_agreement":null},{"id":"W2094392229","doi":"10.1023/a:1009809824225","title":"Beg, Borrow, or Steal: Using Multidisciplinary Approaches in Empirical Software Engineering Research","year":2001,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; University of Toronto; University of Victoria; University of New Brunswick","funders":"University of Victoria","keywords":"Computer science","score_opus":0.286019139000558,"score_gpt":0.4116135689242767,"score_spread":0.12559442992371872,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2094392229","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044304375,0.042169582,0.67521834,0.18550539,0.0029469375,0.0006220995,0.00019140934,0.00087615167,0.048165686],"genre_scores_gemma":[0.38262832,0.03758333,0.5345833,0.027385337,0.002790275,0.003105521,0.00020669274,0.000553067,0.011164256],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9657468,0.026521448,0.0010539441,0.0009678382,0.0051508057,0.0005590371],"domain_scores_gemma":[0.85153705,0.13079809,0.0038980662,0.007183637,0.0039746757,0.0026085223],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.034586966,0.0016764913,0.001880866,0.011608665,0.0032651222,0.012144384,0.0026048834,0.007347099,0.0060151136],"category_scores_gemma":[0.1632817,0.0012632615,0.001017947,0.012327093,0.017548194,0.036391042,0.015767436,0.0074565485,0.0015242703],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017214494,0.0002973598,0.008253488,0.0013435216,0.00030989965,0.0004960004,0.019025395,0.0015549989,0.00068431447,0.6183016,0.028312992,0.32124832],"study_design_scores_gemma":[0.00008059506,0.00004584257,0.0025816062,0.0011303782,0.000112008864,0.00022192397,0.0076635326,0.003999398,0.0002988721,0.96397513,0.019817265,0.00007342779],"about_ca_topic_score_codex":0.0010748862,"about_ca_topic_score_gemma":0.003226578,"teacher_disagreement_score":0.96541303,"about_ca_system_score_codex":0.0019842933,"about_ca_system_score_gemma":0.0040158946,"threshold_uncertainty_score":0.18291557},"labels":[],"label_agreement":null},{"id":"W2100925270","doi":"10.1007/s10664-011-9171-y","title":"An exploratory study of the impact of antipatterns on class change- and fault-proneness","year":2011,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":394,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Queen's University","funders":"","keywords":"Odds; Computer science; Machine learning; Logistic regression","score_opus":0.10704019854817991,"score_gpt":0.3282979284401009,"score_spread":0.221257729891921,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100925270","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9993175,0.0000125178385,0.00021764157,0.000025158643,0.0000010560602,0.000013139981,0.00008861933,0.000006691472,0.00031764508],"genre_scores_gemma":[0.99922657,0.000009990315,0.00038082548,0.000013955653,0.000002872399,0.000023805338,0.0000962269,0.00000447281,0.00024136226],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99599624,0.0025647138,0.00018712578,0.0004160717,0.000590083,0.00024577827],"domain_scores_gemma":[0.69253397,0.2725572,0.020063499,0.0073995464,0.0039538457,0.0034918315],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051322803,0.00033518646,0.00029389438,0.0008750966,0.0005571765,0.0007952459,0.001108225,0.00074310496,0.004539121],"category_scores_gemma":[0.0632035,0.00027673985,0.00050901674,0.0010908948,0.0009453421,0.0015443212,0.0008688805,0.0014162591,0.00041090674],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003505899,0.007942014,0.95250803,0.00012446489,0.00033683973,0.00038366194,0.0028377492,0.0018293514,0.0054524704,0.00094664283,0.0003715011,0.023761254],"study_design_scores_gemma":[0.00013548908,0.0042818324,0.9864547,0.000010874791,0.0001302777,0.00019047131,0.0016180152,0.0045029926,0.0017465068,0.0004994943,0.0004092376,0.0000200329],"about_ca_topic_score_codex":0.0022939949,"about_ca_topic_score_gemma":0.0034595632,"teacher_disagreement_score":0.0051322803,"about_ca_system_score_codex":0.00057457964,"about_ca_system_score_gemma":0.00080931,"threshold_uncertainty_score":0.027142406},"labels":[],"label_agreement":null},{"id":"W2102772625","doi":"10.1007/s10664-006-9006-4","title":"Replaying development history to assess the effectiveness of change propagation tools","year":2006,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of Victoria","funders":"","keywords":"Computer science; Dependency (UML); Notice; Source code; Software engineering; Code review; Open source; Empirical research; Software development; Software; Dependency graph; Code (set theory); Change impact analysis; Data science; Static program analysis; Database; Programming language","score_opus":0.11189261393545909,"score_gpt":0.2994290649800117,"score_spread":0.18753645104455263,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2102772625","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96765745,0.00050377287,0.026642712,0.00013878722,0.00007867403,0.00014882808,0.0008356063,0.0019046917,0.002089501],"genre_scores_gemma":[0.9725568,0.00017171253,0.024805926,0.000034428063,0.000027247776,0.00009365482,0.0012570146,0.00013126002,0.00092195754],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969279,0.0013451326,0.00025872912,0.00047825338,0.0008678115,0.00012218599],"domain_scores_gemma":[0.91438156,0.06001898,0.008053178,0.009830995,0.0065079257,0.0012074624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052497666,0.0006043838,0.00052347576,0.0047651757,0.0003767566,0.00079352,0.0008947533,0.00089885364,0.0014100616],"category_scores_gemma":[0.05817827,0.00031820365,0.00039642397,0.0030500728,0.00032535515,0.0015637581,0.00058465824,0.000905736,0.00041433962],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023878426,0.0019992713,0.24384391,0.00088807626,0.0007101395,0.00036523122,0.0024334113,0.058449805,0.026634699,0.0022033236,0.0033467833,0.6567374],"study_design_scores_gemma":[0.00032037153,0.006229353,0.31685722,0.0002258722,0.0005892427,0.0009876804,0.0012124226,0.6195921,0.041462984,0.0050984863,0.007213875,0.00021039754],"about_ca_topic_score_codex":0.0024756817,"about_ca_topic_score_gemma":0.0023300548,"teacher_disagreement_score":0.0052497666,"about_ca_system_score_codex":0.00044704796,"about_ca_system_score_gemma":0.0005297733,"threshold_uncertainty_score":0.027763784},"labels":[],"label_agreement":null},{"id":"W2109156518","doi":"10.1007/s10664-013-9258-8","title":"Bug characteristics in open source software","year":2013,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":232,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Science Foundation","keywords":"Software bug; Computer science; Concurrency; Security bug; Software regression; Linux kernel; Operating system; Software; Source code; Software engineering; Software system; Software security assurance; Software construction","score_opus":0.025828708118874232,"score_gpt":0.2819716111627644,"score_spread":0.2561429030438902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109156518","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99893266,0.0001388409,0.0004897184,0.00004809863,0.0000025925658,0.000005336566,0.0000654107,0.0000150928045,0.000302247],"genre_scores_gemma":[0.99939394,0.000033478784,0.00027067083,0.0000074245154,0.0000045486067,0.000005197378,0.00012062556,0.000014804597,0.00014937719],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99597484,0.0009827387,0.00063633564,0.0005358008,0.0015050089,0.00036519975],"domain_scores_gemma":[0.7644025,0.13662311,0.072840855,0.0068358267,0.014526241,0.0047714706],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0039257198,0.00024745148,0.00032258962,0.006734596,0.00058761856,0.0014316371,0.0006364435,0.0010092532,0.0014879977],"category_scores_gemma":[0.101130396,0.0004430387,0.0005419804,0.00479579,0.0010587752,0.0029916924,0.0013783533,0.001157922,0.0002504782],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012357776,0.00010299069,0.9868183,0.00002856628,0.00003950317,0.0000933248,0.0007917371,0.0005870449,0.0005107227,0.0004694743,0.00018347753,0.0102512585],"study_design_scores_gemma":[0.000008939404,0.00012793357,0.9945679,0.000028121292,0.00003200799,0.00034225095,0.0008201026,0.0025267939,0.00026563898,0.0010049215,0.00026045236,0.000015007193],"about_ca_topic_score_codex":0.0037961216,"about_ca_topic_score_gemma":0.006649665,"teacher_disagreement_score":0.99607426,"about_ca_system_score_codex":0.00093364954,"about_ca_system_score_gemma":0.00084224535,"threshold_uncertainty_score":0.02076149},"labels":[],"label_agreement":null},{"id":"W2109942136","doi":"10.1007/s10664-008-9104-6","title":"A study of the non-linear adjustment for analogy based software cost estimation","year":2009,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Saskatchewan; University of Wisconsin-Madison","keywords":"Categorical variable; Analogy; Flexibility (engineering); Computer science; Software; Estimation; Artificial intelligence; Artificial neural network; Machine learning; Data mining; Statistics; Mathematics; Engineering","score_opus":0.035501370270765505,"score_gpt":0.320520460559485,"score_spread":0.2850190902887195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109942136","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035159968,0.00051997515,0.95855,0.00050037197,0.000088651024,0.000075555174,0.000031784883,0.00024266173,0.004831104],"genre_scores_gemma":[0.68609124,0.0003858065,0.30589908,0.0001916173,0.00011551115,0.00011301863,0.0001029605,0.00024172482,0.0068590697],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99540544,0.0029443156,0.00014279233,0.00052336045,0.0008295633,0.00015455589],"domain_scores_gemma":[0.9700991,0.02481494,0.0012163578,0.0023495746,0.001368428,0.00015154356],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005097798,0.00050176005,0.0006935793,0.000860425,0.00052389153,0.0014218342,0.0027969424,0.00094254775,0.0061103893],"category_scores_gemma":[0.07071627,0.00048785386,0.00087550655,0.0022467843,0.0009844913,0.003252907,0.0012339397,0.0024260634,0.00049672223],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024447666,0.00025242683,0.0061567025,0.00042054808,0.00020593038,0.00022558746,0.0005378405,0.26642486,0.004636369,0.38619485,0.002366531,0.33233383],"study_design_scores_gemma":[0.000020011763,0.00012290193,0.0033019164,0.000027221266,0.000050299397,0.00013524215,0.000078174206,0.89791447,0.0016461434,0.09288818,0.0037771077,0.000038271457],"about_ca_topic_score_codex":0.0037545753,"about_ca_topic_score_gemma":0.002800257,"teacher_disagreement_score":0.0061103893,"about_ca_system_score_codex":0.0010721005,"about_ca_system_score_gemma":0.0010012792,"threshold_uncertainty_score":0.026960015},"labels":[],"label_agreement":null},{"id":"W2110915013","doi":"10.1007/s10664-012-9205-0","title":"Studying the impact of social interactions on software quality","year":2012,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Software quality; Software; Quality (philosophy); Source code; Software metric; Data science; Software development; Code review; Software engineering; Data mining","score_opus":0.08293084537093437,"score_gpt":0.3962824852350411,"score_spread":0.31335163986410675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2110915013","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9965938,0.00012468782,0.00068678445,0.00027020063,0.0000055506644,0.0000052281,0.000019217458,0.000005888547,0.002288643],"genre_scores_gemma":[0.9993734,0.000055617074,0.00021802832,0.000017977663,0.000009841717,0.0000043816776,0.00001716537,0.0000048436223,0.00029882498],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9956318,0.0027679906,0.0001350225,0.00028081724,0.0008086485,0.0003757444],"domain_scores_gemma":[0.8425017,0.13031778,0.01634014,0.0029258172,0.0041679693,0.0037466197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041981237,0.00030511495,0.00024752322,0.0013060662,0.0008494249,0.001747814,0.000550904,0.00078796735,0.004425776],"category_scores_gemma":[0.058076896,0.00022837728,0.00041355888,0.0013572081,0.0009903942,0.0022202937,0.0010033867,0.0012496255,0.00031555814],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011258983,0.0029971318,0.8747747,0.00028978902,0.00077292987,0.00047300482,0.005456499,0.009676655,0.0062545342,0.010795258,0.0013962416,0.085987434],"study_design_scores_gemma":[0.00014068307,0.0024199823,0.92610264,0.00008865504,0.00043485826,0.00023379846,0.01089483,0.040515617,0.004218497,0.012130919,0.0027369638,0.00008265624],"about_ca_topic_score_codex":0.0045814714,"about_ca_topic_score_gemma":0.0071206293,"teacher_disagreement_score":0.0045814714,"about_ca_system_score_codex":0.0011823182,"about_ca_system_score_gemma":0.00084831275,"threshold_uncertainty_score":0.022202015},"labels":[],"label_agreement":null},{"id":"W2126432237","doi":"10.1023/a:1011930801390","title":"Is it Ethical to Evaluate Web-based Learning Tools using Students?","year":2001,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Usability and User Interface Design","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science","score_opus":0.12063872378271584,"score_gpt":0.3809342550198918,"score_spread":0.26029553123717597,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2126432237","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.916782,0.00093846675,0.0067425673,0.041158594,0.0005604675,0.00016669292,0.00003478603,0.000031195053,0.033585206],"genre_scores_gemma":[0.9914426,0.0002538482,0.0015721434,0.0052163694,0.00008651886,0.000115790266,0.000016776958,0.00002245414,0.0012734465],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.8914477,0.060208548,0.005427604,0.0019548882,0.036833663,0.0041275998],"domain_scores_gemma":[0.5883648,0.24896148,0.041143976,0.014606698,0.08373876,0.023184305],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.061756574,0.00031142682,0.0006156065,0.002180248,0.004220222,0.0115274275,0.0016684652,0.0046850676,0.0029038687],"category_scores_gemma":[0.32512644,0.00031330742,0.00050102884,0.0014534405,0.010936425,0.006646013,0.0041313856,0.0046529896,0.0010965995],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076049333,0.004429016,0.5700376,0.0005398067,0.0003000358,0.000545291,0.115934685,0.00065782556,0.003256665,0.069640964,0.014026118,0.21987157],"study_design_scores_gemma":[0.00046630387,0.0039901366,0.26832893,0.0029483074,0.00034218238,0.0019291572,0.51867807,0.0065614576,0.012371555,0.122969344,0.061111122,0.00030346672],"about_ca_topic_score_codex":0.0013873727,"about_ca_topic_score_gemma":0.0022898712,"teacher_disagreement_score":0.061756574,"about_ca_system_score_codex":0.0034745634,"about_ca_system_score_gemma":0.0074053025,"threshold_uncertainty_score":0.32660383},"labels":[],"label_agreement":null},{"id":"W2142920827","doi":"10.1007/s10664-015-9398-0","title":"Introduction to the special issue on software maintenance and evolution research","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Software development; Open-source software development; Software engineering; Software analytics; Computer science; World Wide Web; Software; Source code; Open source software; Engineering; Software development process; Operating system","score_opus":0.04403391377686007,"score_gpt":0.31887500522402334,"score_spread":0.2748410914471633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142920827","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00035034976,0.073383965,0.006056894,0.044122063,0.85010237,0.00007683256,0.00081512117,0.00036651152,0.024726002],"genre_scores_gemma":[0.0013895561,0.03429588,0.0018928244,0.01565363,0.8825177,0.000080790116,0.0011702651,0.00049049634,0.06250876],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9975873,0.00039879896,0.00028974365,0.00042968095,0.0010832191,0.0002112722],"domain_scores_gemma":[0.98200315,0.008157221,0.0011253036,0.0012329264,0.0046291985,0.0028521041],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037286298,0.0025504455,0.0035754424,0.008578564,0.0016639109,0.008407955,0.002466691,0.0042861826,0.0981949],"category_scores_gemma":[0.013129992,0.0008513262,0.0020331086,0.005620956,0.0016352867,0.007657638,0.0035198778,0.007751443,0.04855593],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016634018,0.000031546948,0.00010361124,0.00023326873,0.000011552282,0.00003423935,0.00001795817,0.00006909893,0.00013697476,0.0017490452,0.9697807,0.0278154],"study_design_scores_gemma":[0.000009301798,0.00004138888,0.0006306006,0.00033121384,0.000016731585,0.00015298845,0.000033398654,0.0001612958,0.00006867837,0.004931184,0.9936067,0.000016476772],"about_ca_topic_score_codex":0.0009198494,"about_ca_topic_score_gemma":0.002191762,"teacher_disagreement_score":0.0981949,"about_ca_system_score_codex":0.0016744925,"about_ca_system_score_gemma":0.0023594429,"threshold_uncertainty_score":0.32849467},"labels":[],"label_agreement":null},{"id":"W2144641356","doi":"10.1007/s10664-006-7552-4","title":"A flexible method for software effort estimation by analogy","year":2006,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":156,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Analogy; Data mining; Computer science; Similarity (geometry); Quality (philosophy); Feature (linguistics); Adaptation (eye); Estimation; Machine learning; Artificial intelligence; Engineering; Image (mathematics)","score_opus":0.017182083426203802,"score_gpt":0.31183334560405335,"score_spread":0.29465126217784954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2144641356","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011775229,0.000030620213,0.9976179,0.000026479755,0.000014770622,0.00002584342,0.000016133996,0.00021433644,0.0008763423],"genre_scores_gemma":[0.11688241,0.00012932476,0.8787065,0.00008074004,0.0000821847,0.00042966942,0.000112616035,0.00018940384,0.0033872125],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9929825,0.0032782767,0.00027478219,0.0012818011,0.0019317886,0.00025088008],"domain_scores_gemma":[0.9895914,0.0061598634,0.00044228247,0.0025484487,0.0011190297,0.00013899474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004920299,0.0011707424,0.001786527,0.002897442,0.0009818826,0.0017615579,0.0030542733,0.0020091352,0.0066309534],"category_scores_gemma":[0.036008995,0.00081021246,0.0016671956,0.0033140557,0.0012161754,0.0048909215,0.00394892,0.0030213988,0.0021221086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014889764,0.0001929295,0.0024013256,0.00019595498,0.00014489466,0.00019417822,0.00038189662,0.07221169,0.0067206277,0.31727135,0.0030785876,0.5970577],"study_design_scores_gemma":[0.00006249863,0.00019388471,0.0018631076,0.00006836536,0.00006941561,0.0004041722,0.000061976905,0.71687174,0.0039101006,0.26755834,0.008845812,0.00009059444],"about_ca_topic_score_codex":0.0011324313,"about_ca_topic_score_gemma":0.0009013037,"teacher_disagreement_score":0.0066309534,"about_ca_system_score_codex":0.0006541079,"about_ca_system_score_gemma":0.00097640615,"threshold_uncertainty_score":0.026021361},"labels":[],"label_agreement":null},{"id":"W2145766063","doi":"10.1007/s10664-012-9206-z","title":"Introduction to the Special Issue on Mining Software Repositories in 2010","year":2012,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"European Metrology Programme for Innovation and Research; Universität des Saarlandes; University of Calgary","keywords":"Computer science; Software engineering; Data science; Software; Operating system","score_opus":0.018376988051948446,"score_gpt":0.27463512077081664,"score_spread":0.2562581327188682,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145766063","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002670291,0.071336776,0.046739213,0.097697064,0.7241182,0.0003141597,0.0076147686,0.0026828025,0.046826784],"genre_scores_gemma":[0.009205637,0.056208592,0.024488289,0.036847804,0.5639704,0.00022859809,0.015011961,0.0024237158,0.291615],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99674964,0.00050208595,0.00043880029,0.0006746676,0.0014424913,0.00019232427],"domain_scores_gemma":[0.97942847,0.005717258,0.001265895,0.0018065083,0.008996169,0.0027857563],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004973392,0.0015359275,0.0021767563,0.010011235,0.0015935518,0.007894778,0.0016599202,0.0024976009,0.052198905],"category_scores_gemma":[0.018070837,0.00083187054,0.0012969348,0.0075341803,0.0011126089,0.0077267303,0.003110318,0.0043376265,0.03564903],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003230402,0.00003573131,0.0002864532,0.00023578167,0.000014918198,0.000045093373,0.00002859669,0.0001461371,0.00040996674,0.0012604883,0.92885804,0.06864651],"study_design_scores_gemma":[0.000008965121,0.000050224906,0.00087674183,0.00021410691,0.000016635837,0.0002156467,0.000038596423,0.0005040186,0.0003427758,0.0023258878,0.99538225,0.000024018373],"about_ca_topic_score_codex":0.0027872797,"about_ca_topic_score_gemma":0.008130524,"teacher_disagreement_score":0.052198905,"about_ca_system_score_codex":0.0024714638,"about_ca_system_score_gemma":0.0036921003,"threshold_uncertainty_score":0.17462271},"labels":[],"label_agreement":null},{"id":"W2149695823","doi":"10.1007/s10664-006-9026-0","title":"Empirical evaluation of optimization algorithms when used in goal-oriented automated test data generation techniques","year":2006,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":59,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Test Management Approach; Computer science; Keyword-driven testing; Test data; Software development; Test harness; Software; Software reliability testing; Automation; Test strategy; Non-regression testing; Data mining; Software construction; Software engineering; Engineering; Programming language","score_opus":0.06930117347575526,"score_gpt":0.3370234376690021,"score_spread":0.26772226419324685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2149695823","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8740962,0.0012424221,0.1191417,0.00038127278,0.00004463692,0.0002726076,0.00032874304,0.0015346513,0.002957827],"genre_scores_gemma":[0.9209353,0.00019097347,0.07771669,0.00004668158,0.00001761189,0.00015147091,0.00042716527,0.00020017455,0.0003139954],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9747628,0.017536275,0.00157797,0.0014824348,0.004127826,0.00051268004],"domain_scores_gemma":[0.64706683,0.31571245,0.008952906,0.015896201,0.011543443,0.0008281814],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.017505858,0.0010282947,0.00064870145,0.0020068763,0.00045565434,0.0011852556,0.0014868038,0.0015983437,0.0009996707],"category_scores_gemma":[0.21532741,0.0004180518,0.00048345237,0.0019439888,0.0010329772,0.0024737183,0.0010317129,0.0013593483,0.0002745125],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039592637,0.0042913877,0.0783202,0.00095799385,0.0004498348,0.0001402851,0.0010062009,0.3768694,0.009446044,0.006148707,0.002845467,0.51556516],"study_design_scores_gemma":[0.00038087816,0.0018324736,0.022168344,0.00009159689,0.00020512886,0.00018972368,0.00018872917,0.957678,0.013418271,0.0026588137,0.0011486218,0.000039478076],"about_ca_topic_score_codex":0.0018611407,"about_ca_topic_score_gemma":0.002181814,"teacher_disagreement_score":0.9824941,"about_ca_system_score_codex":0.001011379,"about_ca_system_score_gemma":0.001143565,"threshold_uncertainty_score":0.092580914},"labels":[],"label_agreement":null},{"id":"W2200054515","doi":"10.1007/s10664-015-9409-1","title":"On the unreliability of bug severity data","year":2015,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Waterloo","funders":"","keywords":"Computer science; Software bug; Reliability (semiconductor); Eclipse; Software; Data mining; Data science; Programming language","score_opus":0.10725834666028859,"score_gpt":0.32480638545586177,"score_spread":0.21754803879557316,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2200054515","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6203692,0.005044597,0.33046567,0.020437956,0.0007031248,0.0003131246,0.006355788,0.0018721076,0.014438348],"genre_scores_gemma":[0.9758321,0.0004980472,0.01894347,0.0012158997,0.00035172404,0.00009840465,0.0022613965,0.00021118391,0.0005877369],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.8796456,0.07537611,0.0092413025,0.0148765715,0.019089129,0.0017712824],"domain_scores_gemma":[0.10724069,0.80469483,0.0278662,0.0482057,0.01117944,0.0008131831],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.09695513,0.00081707706,0.0016303252,0.009304969,0.0016863894,0.0034945388,0.003583395,0.003928188,0.0029721402],"category_scores_gemma":[0.65337944,0.0014451458,0.0013667003,0.008741851,0.0052410383,0.007333129,0.0031392048,0.0054550176,0.00065433263],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010354852,0.00030238004,0.63106614,0.001294127,0.002069394,0.0017333252,0.0040858597,0.10581302,0.001272804,0.08234027,0.019929111,0.14905816],"study_design_scores_gemma":[0.00036885394,0.00037174617,0.25130937,0.0020633088,0.0009832568,0.0038464963,0.0021487984,0.3349619,0.0046092654,0.3798098,0.019250115,0.0002771238],"about_ca_topic_score_codex":0.006121226,"about_ca_topic_score_gemma":0.0045716045,"teacher_disagreement_score":0.9030449,"about_ca_system_score_codex":0.0023490533,"about_ca_system_score_gemma":0.0019229101,"threshold_uncertainty_score":0.51275384},"labels":[],"label_agreement":null},{"id":"W2214818406","doi":"10.1007/s10664-013-9284-6","title":"Management of community contributions","year":2013,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; Polytechnique Montréal; Queen's University","funders":"","keywords":"Thriving; Linux kernel; Android (operating system); Computer science; Empirical research; Knowledge management; Source code; Open source; Software; Software engineering; World Wide Web; Data science; Operating system","score_opus":0.019567496537484873,"score_gpt":0.27656324933417,"score_spread":0.25699575279668513,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2214818406","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.79553825,0.00200681,0.06628559,0.0066653467,0.0003741346,0.0006960967,0.0003499727,0.0007220975,0.1273618],"genre_scores_gemma":[0.98468614,0.00018732232,0.0049946546,0.0000763982,0.00009144872,0.000070314105,0.00010640883,0.000026395664,0.00976089],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.994086,0.0022007085,0.00019757186,0.0008949307,0.0019079823,0.0007127386],"domain_scores_gemma":[0.96029156,0.011053567,0.006281795,0.005120865,0.009762468,0.0074897343],"candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.00690407,0.0005105418,0.0003659778,0.0037016368,0.0028902376,0.0052446746,0.0018796033,0.0014563348,0.008511367],"category_scores_gemma":[0.044716846,0.00025223216,0.00034567362,0.0021167092,0.0010077889,0.0039630607,0.0033229033,0.0012695845,0.00138334],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064446987,0.0018559289,0.12993334,0.00037040218,0.00021036618,0.0006761121,0.0069104386,0.013082741,0.011880789,0.09935915,0.01730978,0.7177665],"study_design_scores_gemma":[0.00041923596,0.0020439522,0.24407353,0.00064642174,0.00037226148,0.0013189927,0.024281042,0.27107924,0.016052423,0.27895665,0.16049348,0.00026274577],"about_ca_topic_score_codex":0.0028289896,"about_ca_topic_score_gemma":0.0053522834,"teacher_disagreement_score":0.9971098,"about_ca_system_score_codex":0.0022409332,"about_ca_system_score_gemma":0.0047219736,"threshold_uncertainty_score":0.036512613},"labels":[],"label_agreement":null},{"id":"W2343606968","doi":"10.1007/s10664-015-9416-2","title":"The impact of domain knowledge on the effectiveness of requirements engineering activities","year":2016,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Domain (mathematical analysis); Systems engineering; Computer science; Domain engineering; Engineering; Engineering management; Knowledge management; Software development; Operating system; Mathematics","score_opus":0.02712372993167587,"score_gpt":0.3155331697157048,"score_spread":0.2884094397840289,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2343606968","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9867444,0.00049525074,0.0016804041,0.0003822321,0.000012709249,0.00003545371,0.000069132904,0.00004161178,0.010538816],"genre_scores_gemma":[0.99867946,0.00010689718,0.00081161084,0.00003303524,0.0000075168355,0.000011830821,0.000064839565,0.000010815813,0.00027388395],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.97673905,0.01551892,0.0013834778,0.0012112479,0.0041555506,0.0009918085],"domain_scores_gemma":[0.25911364,0.70361805,0.01587114,0.009294926,0.009221457,0.0028808904],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019523952,0.0004509995,0.00037921246,0.0019668655,0.0006060344,0.0028778473,0.0009775325,0.0014443382,0.0029435726],"category_scores_gemma":[0.29758707,0.00030458948,0.0004795678,0.0010878583,0.00097628805,0.003384607,0.0012740714,0.0015562219,0.00040694088],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.009339688,0.009964771,0.4234396,0.0016906499,0.0009880131,0.0008173964,0.003725941,0.06420518,0.017155837,0.008167049,0.0013537619,0.45915216],"study_design_scores_gemma":[0.00070402044,0.007840815,0.84120643,0.0006512025,0.0014529676,0.00086356205,0.0049843933,0.097406544,0.02452833,0.016008131,0.004138869,0.00021477185],"about_ca_topic_score_codex":0.002317965,"about_ca_topic_score_gemma":0.0024367603,"teacher_disagreement_score":0.019523952,"about_ca_system_score_codex":0.0016625143,"about_ca_system_score_gemma":0.0022933455,"threshold_uncertainty_score":0.10325372},"labels":[],"label_agreement":null},{"id":"W2411521827","doi":"10.1007/s10664-016-9439-3","title":"Foreword to the special issue on empirical evidence on software product line engineering","year":2016,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Software product line; Product line; Computer science; Software engineering; Software; Engineering; Manufacturing engineering; Software development; Operating system","score_opus":0.09659113352451718,"score_gpt":0.33783418186220177,"score_spread":0.2412430483376846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2411521827","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00017884975,0.017472632,0.000498099,0.08819028,0.8876093,0.00003500097,0.00042987106,0.00009567478,0.0054903403],"genre_scores_gemma":[0.0012081442,0.010944761,0.00041848022,0.048723727,0.91065085,0.00007309556,0.0008245368,0.00022370441,0.02693276],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99656975,0.00060521165,0.00054303516,0.00046406285,0.0014959803,0.00032203484],"domain_scores_gemma":[0.9456514,0.025175411,0.004397462,0.0023917458,0.017517531,0.004866412],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004602552,0.0023943684,0.0032227754,0.008701277,0.0024905815,0.009144555,0.0029305457,0.010551041,0.066768646],"category_scores_gemma":[0.035554565,0.00088663836,0.001614644,0.0045355265,0.0018566983,0.005835801,0.0029649835,0.010106493,0.045285903],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001559991,0.000015571104,0.00007198165,0.00013212133,0.000008155948,0.00002433324,0.000004573489,0.00001022752,0.0000426808,0.00020090207,0.9938711,0.0056026345],"study_design_scores_gemma":[0.000059694525,0.00007029898,0.0020457148,0.0011762764,0.00006883344,0.00020526287,0.00007282189,0.00016595909,0.00025245405,0.0034559632,0.9923954,0.000031348096],"about_ca_topic_score_codex":0.0015956273,"about_ca_topic_score_gemma":0.0027050546,"teacher_disagreement_score":0.066768646,"about_ca_system_score_codex":0.0020606373,"about_ca_system_score_gemma":0.0033490846,"threshold_uncertainty_score":0.2233634},"labels":[],"label_agreement":null},{"id":"W2417033663","doi":"10.1007/s10664-016-9438-4","title":"License usage and changes: a large-scale study on gitHub","year":2016,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"European Commission; National Science Foundation","keywords":"License; Computer science; Traceability; Commit; Java; Software engineering; Python (programming language); AspectJ; Software; JavaScript; Secure coding; World Wide Web; Empirical research; Software development; Reuse; Computer security; Database; Programming language; Engineering; Software security assurance; Operating system","score_opus":0.027664631968529942,"score_gpt":0.28769272460390743,"score_spread":0.2600280926353775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2417033663","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99818265,0.00007532822,0.00007192691,0.0000690788,0.000002764768,0.000022161988,0.000637134,0.00001661874,0.00092243176],"genre_scores_gemma":[0.9958152,0.00015587873,0.00018499694,0.0001394382,0.000012062509,0.000046226127,0.0024054411,0.000061438324,0.0011792447],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99843735,0.00040886173,0.00009333655,0.00025690565,0.00050596474,0.00029761696],"domain_scores_gemma":[0.9854894,0.0059023383,0.004093383,0.0009999565,0.001755936,0.0017589796],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0013079765,0.00042632245,0.00052962435,0.004043209,0.0014341117,0.0018976216,0.0012248083,0.0009143418,0.0028441984],"category_scores_gemma":[0.0083797835,0.00036175986,0.00048812857,0.0074919616,0.0015587511,0.0026304852,0.0020119988,0.0013577421,0.0012729028],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028715222,0.0013470894,0.96220756,0.0001470661,0.00015534875,0.0010916993,0.0143310595,0.00029923904,0.00081454037,0.00040753867,0.0036058647,0.015305981],"study_design_scores_gemma":[0.00000970698,0.00007769039,0.98693323,0.000030751344,0.00002924388,0.00019546064,0.009949966,0.0005132527,0.00021065355,0.0000575207,0.0019726383,0.000019826122],"about_ca_topic_score_codex":0.08933243,"about_ca_topic_score_gemma":0.12779553,"teacher_disagreement_score":0.9959568,"about_ca_system_score_codex":0.0019591174,"about_ca_system_score_gemma":0.0015804162,"threshold_uncertainty_score":0.17762494},"labels":[],"label_agreement":null},{"id":"W2472751774","doi":"10.1007/s10664-016-9435-7","title":"An empirical study of emergency updates for top android mobile apps","year":2016,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Mobile and Web Applications","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University","funders":"","keywords":"Android (operating system); Mobile apps; Computer science; Empirical research; Operating system; World Wide Web; Statistics","score_opus":0.021899480142806616,"score_gpt":0.31454526076882966,"score_spread":0.29264578062602303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2472751774","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99826044,0.00005206479,0.00007594834,0.000110057095,0.0000027805927,0.000018417135,0.000039333936,0.0000024308567,0.0014384484],"genre_scores_gemma":[0.998992,0.000075506665,0.00013219961,0.00004765762,0.0000055233395,0.0000143487005,0.000073842646,0.0000038839244,0.0006551342],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99808276,0.00078068435,0.00016969863,0.00018460814,0.000526701,0.00025560978],"domain_scores_gemma":[0.8865978,0.083043605,0.016428864,0.0026634277,0.008242968,0.0030234002],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029783768,0.00028424762,0.00022111788,0.0014947177,0.0010329167,0.0020193686,0.00074739277,0.0008582858,0.0041036163],"category_scores_gemma":[0.061121713,0.00036618978,0.00025798182,0.0014527515,0.001033306,0.002859234,0.0012145014,0.0019237367,0.0006593789],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000282848,0.0021169835,0.9608142,0.00010339387,0.000047276393,0.00044924158,0.01574428,0.00012731897,0.0005011093,0.0012124781,0.0008096956,0.017791308],"study_design_scores_gemma":[0.000030451376,0.00070283894,0.9542544,0.000084522886,0.00006675868,0.00043087764,0.039970823,0.0013635866,0.00042525592,0.0003787757,0.0022708254,0.000020933656],"about_ca_topic_score_codex":0.0059228544,"about_ca_topic_score_gemma":0.008500325,"teacher_disagreement_score":0.0059228544,"about_ca_system_score_codex":0.00077890756,"about_ca_system_score_gemma":0.0011497319,"threshold_uncertainty_score":0.015751302},"labels":[],"label_agreement":null},{"id":"W2518473846","doi":"10.1007/s10664-016-9451-7","title":"Naming the pain in requirements engineering","year":2016,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":271,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Queen's University; Eesti Teadusagentuur; Queen's University Belfast","keywords":"Relevance (law); Dependency (UML); Context (archaeology); Status quo; Empirical research; Computer science; Management science; Requirements engineering; Complement (music); Data science; Criticality; Engineering ethics; Knowledge management; Risk analysis (engineering); Engineering; Software; Epistemology; Artificial intelligence; Political science; Business","score_opus":0.025830632765819712,"score_gpt":0.27321335161251814,"score_spread":0.24738271884669844,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2518473846","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036413785,0.06810619,0.2503721,0.5120013,0.007921651,0.00012850833,0.00015170976,0.00047607272,0.124428794],"genre_scores_gemma":[0.84250665,0.016203487,0.073034115,0.04517174,0.0062310924,0.00034002037,0.00008938541,0.0005846693,0.015838806],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.9438267,0.045219976,0.001686322,0.0018668403,0.0062259384,0.0011743048],"domain_scores_gemma":[0.8050637,0.1648976,0.0053087627,0.011207344,0.011227108,0.0022955264],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.038415387,0.0008819414,0.001262086,0.004851465,0.0051323455,0.010880279,0.0024867225,0.007969784,0.0061967415],"category_scores_gemma":[0.16936427,0.0009413918,0.0005937642,0.0057859877,0.057805777,0.03677944,0.006046708,0.016460206,0.0010876609],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000042689873,0.000038370592,0.0006624222,0.00027330287,0.000013482222,0.00007325925,0.0082301255,0.00038806107,0.00035425846,0.94293106,0.012862086,0.03413098],"study_design_scores_gemma":[0.000038509294,0.000058875157,0.0009682866,0.0009907824,0.000012154651,0.0001387392,0.008025545,0.0011638746,0.00040306797,0.9276366,0.060516022,0.00004744482],"about_ca_topic_score_codex":0.0030523136,"about_ca_topic_score_gemma":0.0036919499,"teacher_disagreement_score":0.038415387,"about_ca_system_score_codex":0.0046316083,"about_ca_system_score_gemma":0.0040994305,"threshold_uncertainty_score":0.20316243},"labels":[],"label_agreement":null},{"id":"W2531425405","doi":"10.1007/s10664-016-9456-2","title":"Which log level should developers choose for a new logging statement?","year":2016,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University","funders":"","keywords":"Statement (logic); Logging; Computer science; Brier score; Leverage (statistics); Login; Database; Data mining; Information retrieval; Operating system; Machine learning; Forestry","score_opus":0.07458787435700132,"score_gpt":0.30730650497298284,"score_spread":0.23271863061598153,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2531425405","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96164674,0.00036159603,0.013993643,0.0096230395,0.000122808,0.00015349638,0.0007853953,0.00065669033,0.012656613],"genre_scores_gemma":[0.9881621,0.00011913302,0.008767188,0.0005837741,0.000035528363,0.00004485624,0.00018435257,0.00014352395,0.0019595071],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9958662,0.001645915,0.00039383845,0.0006092099,0.0010353643,0.00044953663],"domain_scores_gemma":[0.9215124,0.048644632,0.010102765,0.0052207275,0.009682205,0.0048372718],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.008143733,0.0002944368,0.00029194122,0.000998745,0.0005188838,0.0016982082,0.0006490369,0.0012542605,0.0047327895],"category_scores_gemma":[0.09121022,0.00042321492,0.00028665777,0.0005140971,0.00069981656,0.0030731747,0.0006072848,0.001392272,0.0018382584],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019286288,0.0011831686,0.6853866,0.00048213697,0.00018236527,0.00089782634,0.0035106698,0.0018646386,0.016877623,0.004213073,0.01807809,0.26539522],"study_design_scores_gemma":[0.00063216744,0.0020665147,0.8243482,0.000847475,0.00056960434,0.0020968427,0.024094021,0.035145853,0.037283238,0.03256024,0.03995826,0.0003976187],"about_ca_topic_score_codex":0.0022115011,"about_ca_topic_score_gemma":0.009467151,"teacher_disagreement_score":0.9918563,"about_ca_system_score_codex":0.0007618146,"about_ca_system_score_gemma":0.0014918853,"threshold_uncertainty_score":0.043068707},"labels":[],"label_agreement":null},{"id":"W2534933448","doi":"10.1007/s10664-016-9467-z","title":"Towards just-in-time suggestions for log changes","year":2016,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University","funders":"","keywords":"Commit; Computer science; Logging; Random forest; Set (abstract data type); Directory; Source code; Process (computing); Machine learning; Data mining; Data science; Database; Artificial intelligence; Programming language; Operating system","score_opus":0.023848696009115034,"score_gpt":0.2737739051977656,"score_spread":0.24992520918865058,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2534933448","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07856402,0.001003655,0.8317036,0.017577197,0.0009555122,0.0011337874,0.0014056053,0.046215102,0.02144154],"genre_scores_gemma":[0.3262251,0.00025290303,0.6619252,0.0011799488,0.00026430012,0.00034741883,0.0013127325,0.0018436296,0.006648803],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.959419,0.024672158,0.0019523429,0.003548679,0.009324773,0.0010831326],"domain_scores_gemma":[0.7762057,0.13946699,0.011776331,0.031091558,0.036061715,0.0053977217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023719354,0.0022589667,0.0014239644,0.0032640174,0.0025686875,0.00862879,0.0057011074,0.006374546,0.026861772],"category_scores_gemma":[0.22469547,0.0017360481,0.0010029848,0.0015761234,0.0014756473,0.01373593,0.0048594577,0.0053128754,0.0109468745],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035152554,0.0025409637,0.02447411,0.0014905651,0.00023464096,0.0013368216,0.007051249,0.016686225,0.028418312,0.032647118,0.076991454,0.8046133],"study_design_scores_gemma":[0.0014942865,0.0018074592,0.020637562,0.0012915011,0.00041143462,0.0015221719,0.015421555,0.6136001,0.042407118,0.15319318,0.14748748,0.0007262007],"about_ca_topic_score_codex":0.002659793,"about_ca_topic_score_gemma":0.005934548,"teacher_disagreement_score":0.026861772,"about_ca_system_score_codex":0.0015827768,"about_ca_system_score_gemma":0.006489072,"threshold_uncertainty_score":0.12544143},"labels":[],"label_agreement":null},{"id":"W2536305519","doi":"10.1007/s10664-016-9462-4","title":"Multi-objective reverse engineering of variability-safe feature models based on code dependencies of system variants","year":2016,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"Austrian Science Fund","keywords":"Reverse engineering; Exploit; Computer science; Software product line; Software; Feature engineering; Source code; Feature (linguistics); Software engineering; Data mining; Consolidation (business); Software system; Machine learning; Artificial intelligence; Software development; Programming language","score_opus":0.025584771873456095,"score_gpt":0.2587096846354994,"score_spread":0.23312491276204333,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2536305519","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1924611,0.00009464167,0.80473703,0.00009928677,0.000017849296,0.000067673864,0.00006402566,0.00060577446,0.0018526273],"genre_scores_gemma":[0.8607123,0.000054679414,0.1379356,0.00002356348,0.0000063299035,0.000062434534,0.0001279091,0.00016915704,0.00090795726],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99823195,0.00059386337,0.00007056443,0.00023514299,0.0006829416,0.0001855678],"domain_scores_gemma":[0.9929703,0.004024583,0.00091594417,0.0011992027,0.00077362574,0.00011624894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002370166,0.00088155287,0.0007366521,0.000891601,0.00032842017,0.0009466706,0.0011145333,0.0006017129,0.0009833374],"category_scores_gemma":[0.008970257,0.00057716627,0.0013333204,0.00044854687,0.0008298545,0.0013426341,0.0011076228,0.0012542253,0.00013312441],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000065248285,0.00007602798,0.0020199227,0.00004675611,0.000048133006,0.000091655755,0.00006141118,0.9548068,0.005254285,0.008413845,0.00013062973,0.028985305],"study_design_scores_gemma":[0.0000034680595,0.000023274582,0.0001965889,0.0000031656555,0.000012103575,0.000011981802,0.000007789446,0.9947784,0.0013849495,0.003506627,0.000068508794,0.0000032809291],"about_ca_topic_score_codex":0.003411224,"about_ca_topic_score_gemma":0.0050999173,"teacher_disagreement_score":0.003411224,"about_ca_system_score_codex":0.00095046195,"about_ca_system_score_gemma":0.0013758268,"threshold_uncertainty_score":0.012534797},"labels":[],"label_agreement":null},{"id":"W2543971965","doi":"10.1007/s10664-016-9452-6","title":"Review participation in modern code review","year":2016,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":116,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; McGill University","funders":"","keywords":"Software quality; Code review; Android (operating system); Open source software; Computer science; Process (computing); Source code; Open source; Software; Best practice; Quality (philosophy); Set (abstract data type); Data science; Software development; Political science; Operating system","score_opus":0.04737496011315095,"score_gpt":0.34805441343046845,"score_spread":0.3006794533173175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2543971965","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06788662,0.19137874,0.025517283,0.3875414,0.044408847,0.0045928066,0.0079069855,0.0018798423,0.26888746],"genre_scores_gemma":[0.58963215,0.08556865,0.019356158,0.10220861,0.038912006,0.011076773,0.008021806,0.0022755137,0.1429484],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.7297186,0.14960153,0.024533102,0.018250028,0.0680493,0.0098473495],"domain_scores_gemma":[0.2415008,0.40278572,0.073702544,0.054161023,0.18448032,0.043369595],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.16783057,0.0010418796,0.002366232,0.029018877,0.0050770035,0.013710721,0.0041694622,0.009296178,0.029567556],"category_scores_gemma":[0.62272364,0.0012152501,0.0014977419,0.01698516,0.003899182,0.012059562,0.014407983,0.004901427,0.008521236],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010248759,0.00015423223,0.024446221,0.015457166,0.0008522132,0.0007407256,0.018554412,0.00035596476,0.0027654185,0.028246896,0.5953298,0.31207213],"study_design_scores_gemma":[0.000110416535,0.00008550012,0.013006749,0.0053600217,0.00026244702,0.0003223674,0.0014473383,0.00029067337,0.00073126407,0.005720784,0.97260576,0.000056630928],"about_ca_topic_score_codex":0.0035625768,"about_ca_topic_score_gemma":0.009522077,"teacher_disagreement_score":0.8321694,"about_ca_system_score_codex":0.009304824,"about_ca_system_score_gemma":0.038137443,"threshold_uncertainty_score":0.8875835},"labels":[],"label_agreement":null},{"id":"W2559532455","doi":"10.1007/s10664-016-9487-8","title":"Analysis of license inconsistency in large collections of open source projects","year":2016,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Japan Society for the Promotion of Science","keywords":"License; Computer science; Source code; Header; MIT License; Computer security; Software; Open source; Reuse; World Wide Web; Open source software; Resource (disambiguation); Software engineering; Programming language; Engineering; Operating system","score_opus":0.032707972927695436,"score_gpt":0.309570352120259,"score_spread":0.27686237919256357,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2559532455","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99428034,0.00041192118,0.0028627978,0.00011733776,0.000011102479,0.000043746513,0.0013195362,0.00014405536,0.00080909755],"genre_scores_gemma":[0.9850499,0.00020453318,0.0057945545,0.000039518287,0.00002654851,0.00009663141,0.008141516,0.00015377141,0.00049305265],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98369986,0.0051705604,0.0021011904,0.0023766495,0.005946326,0.00070544594],"domain_scores_gemma":[0.78367764,0.14514823,0.026481649,0.023157125,0.018857293,0.0026780665],"candidate_categories":["metaresearch","bibliometrics","open_science"],"consensus_categories":[],"category_scores_codex":[0.011193052,0.00044869655,0.00094146084,0.011125721,0.0023379826,0.003022758,0.0025525484,0.0015618518,0.0011659358],"category_scores_gemma":[0.12520209,0.0008879577,0.0008817926,0.013082663,0.0017320195,0.00411606,0.0035275128,0.0018818097,0.0004844033],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007523961,0.0010303247,0.8948003,0.0006214432,0.0007431725,0.0010888595,0.007189194,0.010667345,0.0044697467,0.0046228473,0.0055380394,0.06847631],"study_design_scores_gemma":[0.000115477385,0.00030992425,0.88815296,0.00020672653,0.00053077465,0.001848567,0.0073611843,0.0754903,0.006427916,0.008828933,0.010577682,0.00014962222],"about_ca_topic_score_codex":0.009044727,"about_ca_topic_score_gemma":0.010536984,"teacher_disagreement_score":0.99744743,"about_ca_system_score_codex":0.0016899647,"about_ca_system_score_gemma":0.002261096,"threshold_uncertainty_score":0.05919522},"labels":[],"label_agreement":null},{"id":"W2559885217","doi":"10.1007/s10664-017-9512-6","title":"Curating GitHub for engineered software projects","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":345,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Software; Classifier (UML); Software engineering; Software development; Software bug; Software metric; Data mining; Machine learning; Data science; Software quality; Artificial intelligence; Programming language","score_opus":0.05729360034467611,"score_gpt":0.325474775312802,"score_spread":0.2681811749681259,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2559885217","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18679577,0.0020066497,0.5090892,0.0035381946,0.0012920207,0.0016227787,0.009279256,0.21141958,0.07495655],"genre_scores_gemma":[0.2968762,0.0013374245,0.58969104,0.00089080556,0.0002209033,0.00085102,0.023865888,0.048133526,0.038133223],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99223995,0.002226207,0.00042161884,0.0009854302,0.0035837567,0.0005430517],"domain_scores_gemma":[0.96813565,0.00811609,0.002003833,0.013489707,0.007088271,0.0011663334],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005722354,0.0015723514,0.00075341243,0.0062042875,0.0019351395,0.003147876,0.0018914911,0.0013301888,0.011715393],"category_scores_gemma":[0.056724243,0.0009064901,0.0015661491,0.0027672101,0.0010089177,0.004165006,0.008731061,0.0020568154,0.008287753],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006327671,0.00046045324,0.020304834,0.002400512,0.00022890615,0.002096733,0.005882422,0.005422005,0.030398984,0.024300568,0.18277103,0.7251008],"study_design_scores_gemma":[0.00029496904,0.00087832974,0.032284204,0.0022562635,0.0005327007,0.0038784377,0.005613033,0.1172522,0.08057525,0.08575327,0.67030215,0.00037927183],"about_ca_topic_score_codex":0.004172427,"about_ca_topic_score_gemma":0.010705533,"teacher_disagreement_score":0.011715393,"about_ca_system_score_codex":0.0009637127,"about_ca_system_score_gemma":0.0045226943,"threshold_uncertainty_score":0.039191842},"labels":[],"label_agreement":null},{"id":"W2569512077","doi":"10.1007/s10664-016-9475-z","title":"Group versus individual use of power-only EPMcreate as a creativity enhancement technique for requirements elicitation","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Creativity; Brainstorming; Group (periodic table); Requirements elicitation; Raw data; Affect (linguistics); Computer science; Psychology; Social psychology; Requirements analysis; Artificial intelligence; Communication","score_opus":0.10535207857372979,"score_gpt":0.3668034348591423,"score_spread":0.2614513562854125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2569512077","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9528523,0.00007039989,0.03212373,0.00012297869,0.000036125028,0.00050996395,0.000045546138,0.00021865254,0.014020225],"genre_scores_gemma":[0.9510221,0.000051463237,0.044850014,0.00008321234,0.000017768029,0.00067684613,0.00004309176,0.000060176375,0.0031952958],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9973054,0.0014907205,0.00020714031,0.00038296243,0.0004461473,0.00016762674],"domain_scores_gemma":[0.96609885,0.026607487,0.0016171326,0.003593124,0.0011508424,0.00093256065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035961696,0.000273471,0.00041024975,0.000577549,0.0004657182,0.00069449574,0.00069655414,0.00040523312,0.0078030676],"category_scores_gemma":[0.02406719,0.00014781296,0.00019581037,0.0005120709,0.00042023158,0.0009977501,0.0016947783,0.0005365788,0.0007631741],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0062029515,0.007918605,0.013213221,0.0006124607,0.00013128949,0.00020137746,0.010203003,0.002106315,0.06056689,0.0038374797,0.001317304,0.8936891],"study_design_scores_gemma":[0.006991031,0.089206345,0.3554852,0.0011975259,0.002233244,0.0032536115,0.031085476,0.09221304,0.3206962,0.035732217,0.061396766,0.0005093355],"about_ca_topic_score_codex":0.00018143885,"about_ca_topic_score_gemma":0.0007398831,"teacher_disagreement_score":0.0078030676,"about_ca_system_score_codex":0.00026764072,"about_ca_system_score_gemma":0.0005085066,"threshold_uncertainty_score":0.026103795},"labels":[],"label_agreement":null},{"id":"W2586191823","doi":"10.1007/s10664-017-9499-z","title":"Reengineering legacy applications into software product lines: a systematic mapping","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":143,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"Conselho Nacional de Desenvolvimento Científico e Tecnológico; Austrian Science Fund; Coordenação de Aperfeiçoamento de Pessoal de Nível Superior","keywords":"Business process reengineering; Code refactoring; Computer science; Software product line; Software engineering; Process (computing); Process management; Product (mathematics); Reuse; Software; Data science; Software development; Systems engineering; Engineering; Manufacturing engineering","score_opus":0.05418621084206881,"score_gpt":0.3191469998959,"score_spread":0.2649607890538312,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2586191823","genre_codex":"empirical","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91726375,0.019734023,0.047230545,0.0007007179,0.00003888868,0.0027600685,0.0010193199,0.00015337983,0.011099314],"genre_scores_gemma":[0.88834995,0.016253846,0.09033921,0.00033561225,0.000019078097,0.00091976905,0.0013452178,0.0000684065,0.0023688592],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.98406315,0.0076515228,0.0022692515,0.001707533,0.0038116039,0.00049693394],"domain_scores_gemma":[0.86840516,0.082500376,0.01725858,0.013757412,0.017305039,0.0007734106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017537503,0.0006945649,0.00053808483,0.017765356,0.001651496,0.003015372,0.001521636,0.0010441855,0.0020568196],"category_scores_gemma":[0.08685675,0.00067140866,0.0009919513,0.008671902,0.0018152006,0.0060454095,0.0037421284,0.0012134856,0.00036885188],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021240162,0.0011744098,0.14691047,0.009951476,0.00057792495,0.00060698256,0.04231344,0.0021312854,0.004918297,0.012320481,0.0013944486,0.7774885],"study_design_scores_gemma":[0.00027984625,0.00400961,0.5806679,0.06058864,0.004354247,0.00497355,0.1307382,0.019163424,0.035451382,0.027989872,0.13145344,0.00032985694],"about_ca_topic_score_codex":0.0056143156,"about_ca_topic_score_gemma":0.014868097,"teacher_disagreement_score":0.017765356,"about_ca_system_score_codex":0.0024426978,"about_ca_system_score_gemma":0.011538902,"threshold_uncertainty_score":0.092748284},"labels":[],"label_agreement":null},{"id":"W2600697011","doi":"10.1007/s10664-017-9501-9","title":"Documenting and sharing software knowledge using screencasts","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Documentation; Software; Reputation; Software documentation; Knowledge sharing; Multimedia; World Wide Web; Persona; Best practice; Software development; Human–computer interaction; Knowledge management; Software development process","score_opus":0.0560603512164832,"score_gpt":0.33611254988620826,"score_spread":0.2800521986697251,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2600697011","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.53362054,0.00076823676,0.3248578,0.0023588862,0.0005931195,0.002332334,0.003284391,0.009273689,0.122910894],"genre_scores_gemma":[0.89349365,0.000563433,0.08164782,0.00015365811,0.00015800401,0.000731873,0.001323744,0.0005956376,0.021332162],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.99039286,0.005723055,0.00075251714,0.0007173964,0.001982093,0.00043206068],"domain_scores_gemma":[0.86256343,0.1012711,0.0057967063,0.019692812,0.009023419,0.0016525069],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010094813,0.0007008688,0.0005321012,0.005288167,0.0015744398,0.0065163407,0.0011663778,0.0015343796,0.010096832],"category_scores_gemma":[0.07209842,0.0005356197,0.0003241848,0.0033748318,0.0011136014,0.008256189,0.004008926,0.001784506,0.0031744267],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021397138,0.0011528528,0.026283953,0.0017170563,0.0001278216,0.000873712,0.06898328,0.003120542,0.03112322,0.03507631,0.019919416,0.80948204],"study_design_scores_gemma":[0.001327128,0.004272856,0.09685496,0.0036208658,0.00075041066,0.0014081484,0.08064722,0.06721815,0.12138428,0.10039801,0.52122873,0.00088931876],"about_ca_topic_score_codex":0.0028177598,"about_ca_topic_score_gemma":0.0030819774,"teacher_disagreement_score":0.010096832,"about_ca_system_score_codex":0.0010766755,"about_ca_system_score_gemma":0.0022930903,"threshold_uncertainty_score":0.053387105},"labels":[],"label_agreement":null},{"id":"W2604153767","doi":"10.1007/s10664-017-9510-8","title":"An empirical study of unspecified dependencies in make-based build systems","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; Polytechnique Montréal; McGill University; Queen's University","funders":"","keywords":"Deliverable; Computer science; Software engineering; Software deployment; sync; Software; Systems engineering; Programming language; Engineering","score_opus":0.05014827432941954,"score_gpt":0.3386470064963655,"score_spread":0.28849873216694594,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2604153767","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99773896,0.000019538065,0.0008807025,0.00004083932,0.0000015109692,0.00001340265,0.000030259673,0.000011009263,0.0012637179],"genre_scores_gemma":[0.998966,0.000015672107,0.0006164862,0.000008614969,0.0000014479649,0.0000107544565,0.00006183406,0.000006861847,0.0003123065],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9977513,0.0007990406,0.00015080557,0.00024819048,0.00077637716,0.00027433984],"domain_scores_gemma":[0.867914,0.10186554,0.013511287,0.009867588,0.004939627,0.0019018602],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035449464,0.00026431622,0.00020275658,0.0012112268,0.0014893522,0.0011468006,0.0008683111,0.0007290234,0.0033780248],"category_scores_gemma":[0.05806178,0.0004041924,0.00022594762,0.0013648564,0.0015614786,0.0029035306,0.0013569619,0.0015715716,0.00031879087],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070468,0.0029655052,0.89507616,0.0002098973,0.00008975249,0.00074244227,0.01903222,0.010622886,0.0043431944,0.020286635,0.0009216318,0.045004994],"study_design_scores_gemma":[0.000092074886,0.0010725505,0.9195574,0.000103215796,0.000106780324,0.0007330982,0.01630853,0.038312912,0.0060347086,0.01184468,0.0057740286,0.000060067774],"about_ca_topic_score_codex":0.005534154,"about_ca_topic_score_gemma":0.012360718,"teacher_disagreement_score":0.005534154,"about_ca_system_score_codex":0.0013851521,"about_ca_system_score_gemma":0.0017186211,"threshold_uncertainty_score":0.018747687},"labels":[],"label_agreement":null},{"id":"W2604794021","doi":"10.1007/s10664-017-9514-4","title":"What do developers search for on the web?","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":177,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of British Columbia","funders":"Ministry of Science and Technology of the People's Republic of China; National Natural Science Foundation of China; Baidu","keywords":"Computer science; Debugging; World Wide Web; Reuse; Software bug; Software; Search engine optimization; Application programming interface; Information retrieval; Software engineering; Search engine; Data science; Programming language; Engineering","score_opus":0.06330645191543763,"score_gpt":0.3294800053743171,"score_spread":0.2661735534588795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2604794021","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90157074,0.006162768,0.003219423,0.020083899,0.00013232313,0.00008436178,0.0010704113,0.00023261647,0.06744352],"genre_scores_gemma":[0.9887983,0.0018903362,0.0015675548,0.000937304,0.00008712242,0.000021050539,0.0004256868,0.00013696482,0.006135672],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9953418,0.0018379434,0.00025831797,0.0004364643,0.0015863832,0.0005390823],"domain_scores_gemma":[0.9343048,0.041046027,0.01022754,0.0025682992,0.009015879,0.0028375091],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039094575,0.00033830228,0.0005178504,0.004285086,0.0011906287,0.0049642725,0.0007569347,0.002016313,0.007689649],"category_scores_gemma":[0.0618526,0.00035187794,0.00026616175,0.004783501,0.0009729045,0.009995463,0.0010017176,0.0012318451,0.0026782625],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019586203,0.00050033536,0.7339058,0.0008222071,0.00018005597,0.0011457843,0.0145411035,0.00040753445,0.0012729559,0.011933501,0.03618853,0.19890633],"study_design_scores_gemma":[0.00018233275,0.0002790456,0.7477002,0.0022317444,0.00049175555,0.0053750966,0.093205616,0.008510273,0.0047867727,0.03491624,0.102153875,0.00016694353],"about_ca_topic_score_codex":0.012568949,"about_ca_topic_score_gemma":0.020162858,"teacher_disagreement_score":0.012568949,"about_ca_system_score_codex":0.0012720185,"about_ca_system_score_gemma":0.0026512344,"threshold_uncertainty_score":0.02572447},"labels":[],"label_agreement":null},{"id":"W2605594016","doi":"10.1007/s10664-017-9516-2","title":"Data Transformation in Cross-project Defect Prediction","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Transformation (genetics); Computer science; Data mining; Data transformation; Software; Rank (graph theory); Predictive modelling; Reliability engineering; Machine learning; Data warehouse; Mathematics; Engineering","score_opus":0.08894280679107658,"score_gpt":0.37317781031934233,"score_spread":0.28423500352826575,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2605594016","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.50299144,0.0005153908,0.47122696,0.00078314263,0.00024101668,0.00078550103,0.008864594,0.009648155,0.0049437406],"genre_scores_gemma":[0.8421451,0.0001157586,0.14681044,0.00008583666,0.000033772405,0.00047606972,0.008271557,0.0004577289,0.001603731],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98544276,0.0088652335,0.0014171392,0.0017161733,0.0021169046,0.00044187243],"domain_scores_gemma":[0.92028856,0.056201637,0.0026348708,0.015187396,0.0052143284,0.0004731547],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012857264,0.00066875207,0.0007455603,0.003691411,0.0006273522,0.0018912931,0.0012301918,0.00083883107,0.0038284208],"category_scores_gemma":[0.087343566,0.0004444223,0.0011693521,0.005860583,0.00073905656,0.002209198,0.0020011885,0.0016991056,0.0027729508],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002275468,0.0008916195,0.20126693,0.00041516073,0.00035385563,0.00038706756,0.000808296,0.018975893,0.0072687143,0.006657164,0.009652003,0.75104785],"study_design_scores_gemma":[0.00038416547,0.0014722005,0.18366134,0.0003631324,0.00046238198,0.0017323985,0.0026289022,0.68349946,0.05852988,0.03975395,0.027351748,0.00016043105],"about_ca_topic_score_codex":0.00304235,"about_ca_topic_score_gemma":0.002480721,"teacher_disagreement_score":0.012857264,"about_ca_system_score_codex":0.00047276204,"about_ca_system_score_gemma":0.0015543049,"threshold_uncertainty_score":0.0679965},"labels":[],"label_agreement":null},{"id":"W2612705982","doi":"10.1007/s10664-017-9522-4","title":"Identifying self-admitted technical debt in open source projects using text mining","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":190,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; Concordia University","funders":"Ministry of Science and Technology of the People's Republic of China; National Natural Science Foundation of China","keywords":"Technical debt; Computer science; Classifier (UML); Source code; Code review; Baseline (sea); Open source; Artificial intelligence; F1 score; Machine learning; Data mining; Natural language processing; Software; Software quality; Software development; Programming language","score_opus":0.07815053665037296,"score_gpt":0.3568263279761991,"score_spread":0.27867579132582615,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2612705982","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9961922,0.00013924683,0.0013053943,0.00016543905,0.000012149558,0.000024086452,0.0011683499,0.000032995325,0.00096015324],"genre_scores_gemma":[0.9939329,0.00012884014,0.0021977734,0.00004973359,0.00003499869,0.00004894513,0.0028231137,0.000016708245,0.0007670712],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9970227,0.00065216847,0.0007276429,0.00044132094,0.0009093334,0.0002468158],"domain_scores_gemma":[0.9268764,0.038814265,0.022415081,0.0024538063,0.007328088,0.002112351],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034723664,0.00026334552,0.00030170282,0.0068802545,0.000713843,0.0016122028,0.0006649135,0.00096129486,0.001028692],"category_scores_gemma":[0.03971322,0.00018601121,0.0002806805,0.005615596,0.00039532126,0.0021524664,0.001209338,0.00082593726,0.0005037529],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001217976,0.00020806324,0.96066344,0.000117136304,0.00004186226,0.00030680888,0.0011632809,0.0003719807,0.0019374135,0.00049447996,0.0014063134,0.03316738],"study_design_scores_gemma":[0.000014233797,0.00011045082,0.97472715,0.0001559473,0.000056431916,0.0005643507,0.0032981082,0.012678953,0.002632536,0.002212856,0.0035181197,0.000030956366],"about_ca_topic_score_codex":0.0023751673,"about_ca_topic_score_gemma":0.004420753,"teacher_disagreement_score":0.0068802545,"about_ca_system_score_codex":0.0005527096,"about_ca_system_score_gemma":0.0009208096,"threshold_uncertainty_score":0.018363893},"labels":[],"label_agreement":null},{"id":"W2612865512","doi":"10.1007/s10664-017-9520-6","title":"An empirical study of the integration time of fixed issues","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; McGill University","funders":"","keywords":"Eclipse; Computer science; Fixed cost; Heuristics; Fixed point; Fixed effects model; Operations research; Engineering; Business; Mathematics; Accounting; Operating system; Statistics","score_opus":0.038382145223893084,"score_gpt":0.3498034475928034,"score_spread":0.3114213023689103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2612865512","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9902497,0.00028535357,0.0027485495,0.00014709546,0.000010486279,0.00003673552,0.00007503449,0.000025885469,0.0064212186],"genre_scores_gemma":[0.9976259,0.00008279391,0.0012671261,0.00002469106,0.000013011212,0.00001561402,0.00010471432,0.00001766712,0.00084852584],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9966343,0.0017143543,0.00022449088,0.00032789007,0.00082121143,0.00027771253],"domain_scores_gemma":[0.6979755,0.24966656,0.031495687,0.0095408065,0.007478574,0.0038429582],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006418106,0.00025319742,0.00028685026,0.0017463124,0.0007926894,0.0022902372,0.00080839975,0.00092066085,0.011582374],"category_scores_gemma":[0.1561751,0.0003197567,0.0002998201,0.002821774,0.0010702132,0.003943597,0.0011473518,0.0027008604,0.0007539744],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003859422,0.006727056,0.7013536,0.00042172044,0.00022034314,0.0008568135,0.013396658,0.0068517854,0.008878953,0.04480534,0.0028065713,0.20982176],"study_design_scores_gemma":[0.00049988466,0.0046566417,0.8676884,0.00023039484,0.00042879867,0.0013994046,0.014606543,0.055430934,0.008212146,0.032576714,0.014144643,0.00012545657],"about_ca_topic_score_codex":0.0018088244,"about_ca_topic_score_gemma":0.0013535467,"teacher_disagreement_score":0.011582374,"about_ca_system_score_codex":0.001006482,"about_ca_system_score_gemma":0.0010141858,"threshold_uncertainty_score":0.038746893},"labels":[],"label_agreement":null},{"id":"W2623177746","doi":"10.1007/s10664-017-9526-0","title":"An exploratory qualitative and quantitative analysis of emotions in issue report comments of open source systems","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Regione Autonoma della Sardegna","keywords":"Sadness; Gratitude; Recall; Psychology; Emotion classification; Cognitive psychology; Anger; Computer science; Social psychology","score_opus":0.10398874082538329,"score_gpt":0.43043162378672245,"score_spread":0.32644288296133916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2623177746","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9828607,0.000072237395,0.007480012,0.00051895087,0.000083381194,0.00046535846,0.0013239437,0.00007217067,0.007123231],"genre_scores_gemma":[0.98588514,0.000113281334,0.006988532,0.0004754896,0.000099416895,0.0013995364,0.00088921224,0.000097064185,0.004052343],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9893714,0.0066205505,0.00053293695,0.0006436246,0.0021397327,0.0006917541],"domain_scores_gemma":[0.8727664,0.09721487,0.009627407,0.0021049741,0.016275879,0.002010511],"candidate_categories":["scholarly_communication","open_science"],"consensus_categories":[],"category_scores_codex":[0.0078058746,0.00041539533,0.00034136648,0.0024300548,0.0022341902,0.0022539254,0.0006016699,0.0010250864,0.0037153738],"category_scores_gemma":[0.059277672,0.00022871637,0.0002748611,0.0018596709,0.0013544628,0.0016916083,0.0022411519,0.0013919751,0.0008320255],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009806837,0.00079640455,0.09284501,0.0013654702,0.00004147845,0.0010392918,0.8097617,0.00034938805,0.03459059,0.0018293693,0.0049591116,0.051441535],"study_design_scores_gemma":[0.000046003373,0.00091552734,0.2687381,0.0005520976,0.000042908494,0.00048139243,0.696398,0.0017933005,0.009093644,0.0013872547,0.020402689,0.00014903671],"about_ca_topic_score_codex":0.0010704206,"about_ca_topic_score_gemma":0.0023943891,"teacher_disagreement_score":0.99939835,"about_ca_system_score_codex":0.0011743462,"about_ca_system_score_gemma":0.00093372073,"threshold_uncertainty_score":0.04128188},"labels":[],"label_agreement":null},{"id":"W2727304061","doi":"10.1007/s10664-017-9531-3","title":"An empirical study of early access games on the Steam platform","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Educational Games and Gamification","field":"Psychology","cited_by":49,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Technische Universiteit Delft","keywords":"Computer science; Empirical research; Popularity; Game Developer; World Wide Web; Multimedia; Game design","score_opus":0.10240645122112182,"score_gpt":0.41855167410589367,"score_spread":0.31614522288477187,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2727304061","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9970823,0.000023345086,0.0001386067,0.00003603575,0.0000029360729,0.000027430951,0.000012605783,0.0000032992743,0.0026734066],"genre_scores_gemma":[0.99775237,0.000050842715,0.00025821442,0.000031118383,0.00000277128,0.000028678262,0.000036618363,0.000006500573,0.0018328768],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9983096,0.0007963032,0.00008251057,0.0001797243,0.0003477348,0.0002841983],"domain_scores_gemma":[0.96749264,0.022243606,0.0030713845,0.0014128005,0.0027389133,0.003040652],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037024105,0.0005045742,0.00047377846,0.0021400016,0.0012964106,0.0026669733,0.0009764787,0.0010117968,0.0064141513],"category_scores_gemma":[0.038487993,0.00046669968,0.00017885376,0.0015190208,0.0018624656,0.0035118535,0.0019956299,0.0022386506,0.0008305004],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022026289,0.04147925,0.7176992,0.00033369884,0.00011093436,0.001078261,0.12640087,0.00080800924,0.005009399,0.024357745,0.00241538,0.07810458],"study_design_scores_gemma":[0.00018251392,0.003497875,0.8616836,0.00019916362,0.000076415505,0.00037817942,0.11175047,0.004421844,0.001956959,0.0060315705,0.009740418,0.000080989674],"about_ca_topic_score_codex":0.0064013666,"about_ca_topic_score_gemma":0.012203688,"teacher_disagreement_score":0.0064141513,"about_ca_system_score_codex":0.0010552271,"about_ca_system_score_gemma":0.0011298056,"threshold_uncertainty_score":0.021457434},"labels":[],"label_agreement":null},{"id":"W2727610827","doi":"10.1007/s10664-017-9529-x","title":"Noise in Mylyn interaction traces and its impact on developers and recommendation systems","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Noise (video); Software; Code (set theory); Source code; Recommender system; Data science; Human–computer interaction; Information retrieval; Software engineering; Data mining; Artificial intelligence; Programming language; Image (mathematics)","score_opus":0.04160546049285644,"score_gpt":0.3472158973043576,"score_spread":0.30561043681150113,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2727610827","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9845536,0.00023140195,0.010460142,0.0009901338,0.000064349246,0.000043691707,0.0007374094,0.00076102343,0.0021582274],"genre_scores_gemma":[0.99529195,0.00004520713,0.0021695688,0.00007928289,0.00002677014,0.000032768945,0.0007510285,0.0001372048,0.0014661205],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98703927,0.0054835663,0.00089533365,0.0020456011,0.0038628005,0.00067339465],"domain_scores_gemma":[0.6546255,0.28459486,0.01659839,0.021577993,0.016757784,0.0058454783],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009496763,0.0004133096,0.0006758367,0.0027989165,0.0011976777,0.0033463512,0.0012990648,0.0020466072,0.003452421],"category_scores_gemma":[0.24338742,0.00072247244,0.0003283895,0.002222286,0.001235216,0.0044198954,0.0021446378,0.002752805,0.0010701469],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032778112,0.0015106754,0.78810424,0.00029535126,0.00042123647,0.00088814384,0.008076981,0.030168442,0.010289967,0.00994516,0.010560904,0.13646097],"study_design_scores_gemma":[0.00017487626,0.0012080522,0.5566933,0.00021027801,0.00030463206,0.0012503298,0.005659446,0.38907915,0.011429545,0.022721289,0.010994276,0.00027482765],"about_ca_topic_score_codex":0.011901148,"about_ca_topic_score_gemma":0.013385512,"teacher_disagreement_score":0.011901148,"about_ca_system_score_codex":0.0017080551,"about_ca_system_score_gemma":0.0016276629,"threshold_uncertainty_score":0.050224304},"labels":[],"label_agreement":null},{"id":"W2738381526","doi":"10.1007/s10664-017-9533-1","title":"EnTagRec ++: An enhanced tag recommendation system for software information sites","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":81,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Ask price; Computer science; Set (abstract data type); Software; Recall; Precision and recall; Information retrieval; World Wide Web; Operating system","score_opus":0.03528399466237,"score_gpt":0.3128422166076763,"score_spread":0.2775582219453063,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2738381526","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04843058,0.0016599514,0.29895142,0.00066140044,0.0005055787,0.0012361998,0.057573564,0.57869756,0.0122838095],"genre_scores_gemma":[0.17896825,0.0012516397,0.6316577,0.0008744651,0.00042500222,0.00084158266,0.1246404,0.008985099,0.05235577],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99859816,0.00024722892,0.00011678067,0.00026940042,0.00065715855,0.00011126765],"domain_scores_gemma":[0.9968401,0.00080898387,0.00022914103,0.0010149949,0.0007411945,0.000365675],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013482538,0.0021364326,0.0018209654,0.0065635904,0.0008174213,0.0017517938,0.0018668283,0.0014645249,0.018292231],"category_scores_gemma":[0.00472981,0.00075558375,0.0010858891,0.0042024353,0.00015550134,0.0025984652,0.0019322314,0.0010509003,0.024486851],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025710496,0.0007895195,0.009116318,0.00094009127,0.0005154988,0.00049185724,0.00022751784,0.003730399,0.031756803,0.0017389634,0.25961584,0.6885062],"study_design_scores_gemma":[0.0010483483,0.0014836121,0.027508007,0.00019274934,0.00084844657,0.0015026416,0.00048476402,0.5298834,0.09586393,0.009831788,0.3305325,0.0008197309],"about_ca_topic_score_codex":0.0101200165,"about_ca_topic_score_gemma":0.024547275,"teacher_disagreement_score":0.018292231,"about_ca_system_score_codex":0.00055885455,"about_ca_system_score_gemma":0.000764681,"threshold_uncertainty_score":0.061193585},"labels":[],"label_agreement":null},{"id":"W2755579216","doi":"10.1007/s10664-017-9547-8","title":"Inference of development activities from interaction with uninstrumented applications","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Documentation; Software engineering; Software development; Generalizability theory; Software; Event (particle physics); Set (abstract data type); Data science; Human–computer interaction; Programming language","score_opus":0.03065809452857621,"score_gpt":0.3057166064988471,"score_spread":0.27505851197027087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2755579216","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8743093,0.00042415998,0.11789103,0.00026864948,0.000029172681,0.00008077405,0.0024312472,0.0013085285,0.003257101],"genre_scores_gemma":[0.9818127,0.00009192041,0.015257395,0.000021725069,0.000017339145,0.000032696593,0.0020427594,0.0000718256,0.0006516234],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9963043,0.001461851,0.0002519086,0.00087004044,0.0008118467,0.00030001398],"domain_scores_gemma":[0.9315521,0.055795178,0.0035078286,0.00633849,0.0021317285,0.0006746345],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037011064,0.0006183217,0.0004929721,0.0028491956,0.000320156,0.0014831094,0.0012898602,0.001248255,0.0020744433],"category_scores_gemma":[0.04844766,0.00063462247,0.0009213761,0.0016545422,0.0005290706,0.0020674022,0.0011602322,0.0015004462,0.0010465083],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019778209,0.0012358959,0.6411399,0.00046736863,0.00052576844,0.0005642706,0.0013748643,0.05861074,0.012683031,0.007614177,0.0024782557,0.2713279],"study_design_scores_gemma":[0.0000897128,0.00039801674,0.24910839,0.000107609056,0.0002760933,0.00044583605,0.0005363302,0.7070461,0.018589886,0.020247027,0.0030922254,0.00006271121],"about_ca_topic_score_codex":0.006502566,"about_ca_topic_score_gemma":0.0074893734,"teacher_disagreement_score":0.006502566,"about_ca_system_score_codex":0.00065255025,"about_ca_system_score_gemma":0.0009375759,"threshold_uncertainty_score":0.01957357},"labels":[],"label_agreement":null},{"id":"W2757905449","doi":"10.1007/s10664-017-9545-x","title":"An exploratory study on assessing the energy impact of logging on Android applications","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Green IT and Sustainability","field":"Engineering","cited_by":49,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Alberta Innovates - Technology Futures; Johns Hopkins University","keywords":"Android (operating system); Logging; Computer science; Exploratory research; Operating system; Forestry; Geography","score_opus":0.027996351793043654,"score_gpt":0.33390954012214097,"score_spread":0.3059131883290973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2757905449","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9981382,0.0000146499015,0.00025941682,0.000028149934,0.0000020469056,0.000079078876,0.00004617384,0.0000074917198,0.0014248665],"genre_scores_gemma":[0.997601,0.000056750203,0.00074933027,0.00004385943,0.000004336708,0.00007020886,0.00008960679,0.000007822059,0.0013772735],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9989724,0.0004062498,0.00005687794,0.00011784484,0.0002774113,0.00016917834],"domain_scores_gemma":[0.97965825,0.01491507,0.0014840813,0.00097377267,0.0024075867,0.00056120515],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0017434533,0.00038850546,0.0002551076,0.0009085321,0.0011831733,0.0009037038,0.0006787773,0.000673247,0.0027231898],"category_scores_gemma":[0.013859438,0.00025384594,0.00028273027,0.0009567345,0.00082516595,0.0016597813,0.0008351651,0.00090818846,0.0004580875],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021765025,0.023622973,0.7339119,0.0014176149,0.00015991139,0.0034356047,0.057787288,0.0045744264,0.032972515,0.0037636925,0.002218462,0.13395919],"study_design_scores_gemma":[0.00013431907,0.011544836,0.8567719,0.00029262577,0.00017771337,0.00084758707,0.09046871,0.010207148,0.01908415,0.0017115934,0.008657347,0.00010208083],"about_ca_topic_score_codex":0.0050862296,"about_ca_topic_score_gemma":0.01188828,"teacher_disagreement_score":0.99825656,"about_ca_system_score_codex":0.0007551609,"about_ca_system_score_gemma":0.00096907536,"threshold_uncertainty_score":0.010113299},"labels":[],"label_agreement":null},{"id":"W2763684087","doi":"10.1007/s10664-017-9551-z","title":"Analyzing a decade of Linux system calls","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Linux kernel; Computer science; Operating system; System call; sysfs; Configfs; Kernel (algebra); GNU/Linux; Set (abstract data type); Source lines of code; Application programming interface; Software engineering; Programming language; Software","score_opus":0.03008028736372726,"score_gpt":0.30459034001024504,"score_spread":0.2745100526465178,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2763684087","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98972416,0.0009902295,0.0014299833,0.00093354774,0.000055550154,0.000009442422,0.0011572784,0.00013416274,0.0055655283],"genre_scores_gemma":[0.9940937,0.0003939316,0.00085829495,0.00015322014,0.00007018516,0.000009039553,0.0021723201,0.000093906754,0.0021553664],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9984029,0.00025915026,0.00010129305,0.00027085075,0.00071132724,0.00025448998],"domain_scores_gemma":[0.97801614,0.011932715,0.0033581026,0.0017640007,0.0038823716,0.0010465841],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018535074,0.0002364228,0.00021711507,0.0031940113,0.0009005204,0.0018283633,0.00068070326,0.0009003548,0.001636374],"category_scores_gemma":[0.024443066,0.00033888427,0.00022805047,0.0040638857,0.00083755,0.0020935694,0.0009509166,0.0015880677,0.0005866552],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00068983604,0.0005964462,0.7072159,0.0003247074,0.0002202102,0.0011163084,0.012993543,0.01203841,0.007311071,0.024222573,0.034359418,0.19891162],"study_design_scores_gemma":[0.000021805658,0.00023135745,0.8840866,0.00023250873,0.00009443003,0.0005874082,0.009572824,0.027069563,0.003839304,0.0057928017,0.06838535,0.00008605983],"about_ca_topic_score_codex":0.025290895,"about_ca_topic_score_gemma":0.037283313,"teacher_disagreement_score":0.025290895,"about_ca_system_score_codex":0.0016852106,"about_ca_system_score_gemma":0.0012001821,"threshold_uncertainty_score":0.050287366},"labels":[],"label_agreement":null},{"id":"W2763776913","doi":"10.1007/s10664-017-9553-x","title":"Empirical study on the discrepancy between performance testing results from virtual and physical environments","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Empirical research; Human–computer interaction; Statistics; Mathematics","score_opus":0.05186541589034044,"score_gpt":0.2904156404946555,"score_spread":0.23855022460431508,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2763776913","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99628407,0.000093555616,0.0019969316,0.00006751205,0.000005398092,0.000008596989,0.000099138,0.000014918377,0.001429937],"genre_scores_gemma":[0.9992932,0.000030975407,0.00040887127,0.000015905467,0.0000034427162,0.000005797388,0.00013050949,0.000007674442,0.00010348311],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98656034,0.0075982367,0.0010406541,0.00095031236,0.0034666548,0.00038373267],"domain_scores_gemma":[0.55219877,0.3899872,0.02617643,0.0111909555,0.018772941,0.0016736963],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012084378,0.0004022687,0.00021279299,0.0019505324,0.0004037913,0.0010103908,0.0012115499,0.0007910489,0.0015503938],"category_scores_gemma":[0.16737348,0.00022996472,0.00025203722,0.0020232352,0.001168139,0.0016603803,0.0011530855,0.0008965732,0.00040303904],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067875587,0.0009807198,0.9544966,0.00017947573,0.00015149852,0.0002686319,0.0061090016,0.0032194695,0.0025734405,0.0017653446,0.000659971,0.028917095],"study_design_scores_gemma":[0.000049726576,0.0014319724,0.9721343,0.000119154705,0.0001152594,0.00093010516,0.0060965624,0.010959809,0.0053151688,0.0013999285,0.0014126276,0.000035287863],"about_ca_topic_score_codex":0.0013999249,"about_ca_topic_score_gemma":0.0014675424,"teacher_disagreement_score":0.012084378,"about_ca_system_score_codex":0.0005257123,"about_ca_system_score_gemma":0.00044372742,"threshold_uncertainty_score":0.063909054},"labels":[],"label_agreement":null},{"id":"W2765212458","doi":"10.1007/s10664-017-9558-5","title":"Understanding the factors for fast answers in technical Q&amp;A websites","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Expert finding and Q&A systems","field":"Computer Science","cited_by":69,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University","funders":"","keywords":"Incentive; Dimension (graph theory); Order (exchange); Quality (philosophy); Computer science; Question answering; World Wide Web; Internet privacy; Business; Information retrieval; Mathematics","score_opus":0.15669093957551167,"score_gpt":0.32453475118650243,"score_spread":0.16784381161099077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2765212458","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95137346,0.00048580245,0.0068647033,0.0045786733,0.000060972026,0.00030681692,0.00027541607,0.00022411464,0.035830013],"genre_scores_gemma":[0.99474734,0.000100980425,0.0025930877,0.00019561178,0.000038003764,0.00004727259,0.00010628628,0.00007048085,0.002100959],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9819357,0.008843511,0.001666587,0.0013124337,0.004225649,0.0020161355],"domain_scores_gemma":[0.31342426,0.5890073,0.04497276,0.013771589,0.029326763,0.009497328],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024830248,0.00036332555,0.0004363427,0.0038824254,0.0024832238,0.009130305,0.0016462103,0.003636398,0.022101425],"category_scores_gemma":[0.34377906,0.0008100117,0.0007931367,0.0026103053,0.0026089002,0.013801231,0.0022639914,0.004663681,0.0029564702],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009263845,0.0027911644,0.84352344,0.0005708533,0.00019260035,0.00088866096,0.015543562,0.002053444,0.00200956,0.03318415,0.007855382,0.09046073],"study_design_scores_gemma":[0.00020832139,0.0006025037,0.91490394,0.000583841,0.00027494022,0.0009060398,0.02585769,0.011137092,0.0026835797,0.030612074,0.012055862,0.00017420735],"about_ca_topic_score_codex":0.012249956,"about_ca_topic_score_gemma":0.012618293,"teacher_disagreement_score":0.024830248,"about_ca_system_score_codex":0.0028244602,"about_ca_system_score_gemma":0.0071783527,"threshold_uncertainty_score":0.13131648},"labels":[],"label_agreement":null},{"id":"W2767795225","doi":"10.1007/s10664-017-9559-4","title":"Are tweets useful in the bug fixing process? An empirical study on Firefox and Chrome","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; World Wide Web; Social media; Process (computing); Empirical research; Security bug; Microblogging; Software; Internet privacy; Computer security","score_opus":0.06260857921942889,"score_gpt":0.36454485204685283,"score_spread":0.30193627282742397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2767795225","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9988537,0.000071142014,0.00009042032,0.00018729985,0.000008509571,0.000012068556,0.000100223995,0.000008767184,0.0006680115],"genre_scores_gemma":[0.998657,0.00007700311,0.00022405262,0.000077904566,0.000019251733,0.0000140501625,0.0002751934,0.000016602839,0.0006388741],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9979722,0.0009561869,0.00010845163,0.00021812164,0.00047124608,0.00027371632],"domain_scores_gemma":[0.9090018,0.069998525,0.0123163285,0.0018066766,0.004293144,0.0025833913],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027399543,0.00032918094,0.00030168326,0.0016547117,0.0012975056,0.00209292,0.00064162485,0.001469494,0.0019238209],"category_scores_gemma":[0.044225994,0.00029885207,0.0002639186,0.001445395,0.00080875953,0.0031976735,0.0007216116,0.001929505,0.0005754585],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008794581,0.001815744,0.9448568,0.0001977052,0.00010171432,0.0006263214,0.01998851,0.00034958214,0.0027441047,0.00076916436,0.0023117836,0.025359066],"study_design_scores_gemma":[0.000054295404,0.0005241237,0.9678115,0.00009776734,0.0001253502,0.00027295083,0.022479633,0.0029883287,0.0015333605,0.00028659913,0.003780279,0.000045687357],"about_ca_topic_score_codex":0.016365439,"about_ca_topic_score_gemma":0.016914064,"teacher_disagreement_score":0.016365439,"about_ca_system_score_codex":0.0010742617,"about_ca_system_score_gemma":0.0009686766,"threshold_uncertainty_score":0.03254038},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"}],"label_agreement":"agree"},{"id":"W2786424616","doi":"10.1007/s10664-018-9595-8","title":"Studying software logging using topic models","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":95,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University","funders":"","keywords":"Logging; Computer science; Software; Software engineering; Operating system; Forestry; Geography","score_opus":0.0777747160628274,"score_gpt":0.31632667881580295,"score_spread":0.23855196275297555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2786424616","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.755968,0.0026501513,0.22799452,0.0024793898,0.00007404296,0.00011724861,0.0002920903,0.00048303817,0.009941475],"genre_scores_gemma":[0.9888076,0.0007088084,0.008644212,0.000068406065,0.00009874165,0.00005865646,0.00026089026,0.00008941415,0.0012631406],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9963007,0.0026723633,0.00012356688,0.00037212027,0.0003120316,0.00021930184],"domain_scores_gemma":[0.8847153,0.10646708,0.0029517377,0.0032276143,0.0016903467,0.0009480038],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007309626,0.0007300656,0.00083487347,0.0025919345,0.0010110938,0.0040203393,0.001142146,0.0015939303,0.0032530588],"category_scores_gemma":[0.060726386,0.0007223399,0.00089385477,0.00410205,0.0010399777,0.01049256,0.001217949,0.0024139315,0.00065260875],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013052016,0.0025179693,0.2996075,0.00107492,0.00082719815,0.0006077637,0.01995619,0.11586348,0.0070619653,0.22078556,0.01067179,0.31972048],"study_design_scores_gemma":[0.000118958225,0.0005209794,0.0523946,0.0002086508,0.00033315222,0.00054835813,0.00903551,0.6887113,0.0032829496,0.23535807,0.009396489,0.00009102552],"about_ca_topic_score_codex":0.0031644837,"about_ca_topic_score_gemma":0.0032836339,"teacher_disagreement_score":0.007309626,"about_ca_system_score_codex":0.0013039203,"about_ca_system_score_gemma":0.0009043756,"threshold_uncertainty_score":0.038657486},"labels":[],"label_agreement":null},{"id":"W2788314565","doi":"10.1007/s10664-018-9601-1","title":"App store mining is not enough for app improvement","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":84,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of Calgary","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Alberta Innovates - Technology Futures","keywords":"App store; Computer science; Mobile apps; Smartphone app; World Wide Web","score_opus":0.025491394227604578,"score_gpt":0.2673284083448976,"score_spread":0.241837014117293,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2788314565","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5167426,0.016124174,0.23136082,0.12728284,0.002833852,0.0013053415,0.011454697,0.02215057,0.07074511],"genre_scores_gemma":[0.83168393,0.003511346,0.13413046,0.007935736,0.0014973421,0.00035038442,0.0065745763,0.00111683,0.013199375],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98703253,0.0031346276,0.0010855063,0.0017935178,0.0062297615,0.00072388345],"domain_scores_gemma":[0.8506922,0.0712537,0.010458997,0.026319556,0.0388659,0.0024095746],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010762944,0.0020375233,0.0023736684,0.0061367536,0.0016940823,0.005938727,0.0032095904,0.0026059092,0.007900404],"category_scores_gemma":[0.100259855,0.0011484163,0.0015524145,0.004262989,0.0016109247,0.019723885,0.0020485576,0.004213605,0.008334548],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036104364,0.0012629754,0.18146738,0.0012182202,0.0005818627,0.00023315553,0.000633903,0.003864944,0.004481129,0.007210069,0.08108281,0.7176025],"study_design_scores_gemma":[0.00024143906,0.0018836479,0.21845186,0.003634074,0.0018540451,0.0026608352,0.0066845235,0.3296129,0.030522047,0.18053176,0.223533,0.00038984258],"about_ca_topic_score_codex":0.005084589,"about_ca_topic_score_gemma":0.008800792,"teacher_disagreement_score":0.010762944,"about_ca_system_score_codex":0.001230145,"about_ca_system_score_gemma":0.0040162057,"threshold_uncertainty_score":0.056920588},"labels":[],"label_agreement":null},{"id":"W2790537587","doi":"10.1007/s10664-018-9603-z","title":"Studying and detecting log-related issues","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":64,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Logging; Computer science; Statement (logic); Data science; Software; Scale (ratio); Root (linguistics); Open source; Data mining; Operating system","score_opus":0.019018094875514664,"score_gpt":0.27283382348902824,"score_spread":0.2538157286135136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2790537587","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95912564,0.00056059135,0.033381958,0.0007705812,0.000056631972,0.00023367096,0.0005018066,0.00072442717,0.0046447027],"genre_scores_gemma":[0.99023074,0.00023618813,0.008014972,0.00007473155,0.00003553293,0.000045798864,0.0004206203,0.000046132016,0.0008952811],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99600804,0.0015124303,0.00040311186,0.00037074956,0.0014503823,0.00025534126],"domain_scores_gemma":[0.8621967,0.09444061,0.023393666,0.008352042,0.010257612,0.0013593476],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045733587,0.00050636847,0.00032134677,0.004519449,0.0006531283,0.001729528,0.00087863876,0.0008965258,0.0021349033],"category_scores_gemma":[0.09439272,0.00036867222,0.00028014154,0.0023646308,0.0005551999,0.0041514486,0.00096916227,0.0012614466,0.000455368],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036733682,0.0009988383,0.8248312,0.00040263528,0.00007338424,0.00065279414,0.0018134122,0.003995639,0.0071034194,0.0036712089,0.0026319146,0.15345818],"study_design_scores_gemma":[0.000089862755,0.0014116439,0.773449,0.000425433,0.00024558176,0.002632455,0.0073697893,0.15582652,0.027150676,0.019897483,0.01141007,0.000091499],"about_ca_topic_score_codex":0.0020754396,"about_ca_topic_score_gemma":0.0031669438,"teacher_disagreement_score":0.0045733587,"about_ca_system_score_codex":0.00062901777,"about_ca_system_score_gemma":0.0017189018,"threshold_uncertainty_score":0.024186552},"labels":[],"label_agreement":null},{"id":"W2792451387","doi":"10.1007/s10664-018-9600-2","title":"On the correctness of electronic documents: studying, finding, and localizing inconsistency bugs in PDF readers and files","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Engineering and Physical Sciences Research Council; Natural Sciences and Engineering Research Council of Canada; Microsoft Research","keywords":"Correctness; Computer science; Information retrieval; World Wide Web; Programming language","score_opus":0.015947982980012043,"score_gpt":0.2682681163052177,"score_spread":0.25232013332520564,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2792451387","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94119126,0.0018337771,0.051802687,0.00045531782,0.00005416436,0.00028327393,0.0010623771,0.0016776177,0.0016395717],"genre_scores_gemma":[0.9159771,0.000785875,0.07919738,0.00016313363,0.00005718234,0.00014540934,0.002306357,0.00036672252,0.0010008431],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98537874,0.0036151141,0.0024434237,0.0026929916,0.005409294,0.00046046966],"domain_scores_gemma":[0.74942195,0.17262562,0.039030805,0.013015171,0.024784153,0.0011222813],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008607044,0.00067945325,0.00089498307,0.012322072,0.0014180905,0.0027579411,0.0017066394,0.0015696745,0.0006802518],"category_scores_gemma":[0.13156304,0.000743259,0.00049580313,0.0075353566,0.001867279,0.005455444,0.0020932935,0.0011245784,0.00043415575],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005727429,0.00048714428,0.5883526,0.0022708497,0.00023693359,0.00284503,0.019531025,0.006177117,0.026316656,0.0026139054,0.0049636117,0.34563246],"study_design_scores_gemma":[0.00015475028,0.0009017375,0.68624467,0.0010654881,0.0005584673,0.010793585,0.015407011,0.1210099,0.1253408,0.0063651362,0.031782523,0.00037598776],"about_ca_topic_score_codex":0.0056525175,"about_ca_topic_score_gemma":0.007411249,"teacher_disagreement_score":0.012322072,"about_ca_system_score_codex":0.0012957711,"about_ca_system_score_gemma":0.0015201466,"threshold_uncertainty_score":0.045518935},"labels":[],"label_agreement":null},{"id":"W2793175118","doi":"10.1007/s10664-018-9604-y","title":"Studying the consistency of star ratings and the complaints in 1 &amp; 2-star user reviews for top free cross-platform Android and iOS apps","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Android (operating system); Computer science; Mobile apps; World Wide Web; Revenue; Star (game theory); App store; Download; User experience design; Consistency (knowledge bases); Human–computer interaction; Operating system; Artificial intelligence","score_opus":0.05466891975716319,"score_gpt":0.32305640697843746,"score_spread":0.26838748722127426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2793175118","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98827994,0.00065651897,0.0032758103,0.0005360902,0.00016306601,0.00020360152,0.0013643061,0.00012777623,0.0053928327],"genre_scores_gemma":[0.9942525,0.00015359167,0.001951582,0.0001938805,0.000085938256,0.00020872675,0.00167971,0.000065222695,0.0014087451],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9356899,0.019501472,0.012108945,0.0053846436,0.025805965,0.0015089944],"domain_scores_gemma":[0.46141055,0.29805914,0.08257044,0.01932023,0.13409822,0.0045414804],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04069731,0.00033827455,0.0008852299,0.0071166665,0.0011923113,0.0029582684,0.00097597146,0.0010588277,0.0014785215],"category_scores_gemma":[0.30747947,0.00040413672,0.0011444237,0.00497898,0.0011916097,0.0034557625,0.003042175,0.0011590652,0.00082001224],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010462541,0.00020399087,0.92810506,0.0008216083,0.0007929611,0.00014077472,0.016580047,0.0004384766,0.0030656029,0.00064199726,0.005645695,0.042517498],"study_design_scores_gemma":[0.00002870785,0.00055063434,0.98391336,0.0001299955,0.00019528673,0.0002256902,0.006226279,0.0022095703,0.0014475434,0.00024665872,0.0047473307,0.000078939556],"about_ca_topic_score_codex":0.0027867043,"about_ca_topic_score_gemma":0.005467275,"teacher_disagreement_score":0.04069731,"about_ca_system_score_codex":0.0013588689,"about_ca_system_score_gemma":0.001264213,"threshold_uncertainty_score":0.21523052},"labels":[],"label_agreement":null},{"id":"W2793227253","doi":"10.1007/s10664-017-9592-3","title":"ProMeTA: a taxonomy for program metamodels in program reverse engineering","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Japan Society for the Promotion of Science","keywords":"Metamodeling; Taxonomy (biology); Computer science; Reuse; Software engineering; Program comprehension; Systems engineering; Orthogonality; Artificial intelligence; Engineering; Programming language; Software; Software system","score_opus":0.0615456604311363,"score_gpt":0.3297100329207495,"score_spread":0.2681643724896132,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2793227253","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0069016726,0.0036227345,0.9711057,0.0019466772,0.00016043348,0.0013990763,0.001847621,0.0022119528,0.010804081],"genre_scores_gemma":[0.027015118,0.0032058232,0.9613904,0.00045338966,0.000059051134,0.0021253047,0.0033781363,0.0003565154,0.0020161993],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98575145,0.005334822,0.0036301373,0.001623399,0.0030895139,0.0005706397],"domain_scores_gemma":[0.9763856,0.01026515,0.0031494687,0.0046408786,0.0048208954,0.0007379487],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013787038,0.0017147232,0.0011953032,0.017833598,0.0032277298,0.007226504,0.0031681242,0.0030486898,0.0031974595],"category_scores_gemma":[0.026835347,0.0012820405,0.0036632998,0.015955947,0.004181694,0.017243909,0.0053956364,0.005018899,0.0015417153],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000099173085,0.00019467673,0.0068421955,0.0032813877,0.00012693822,0.0005599754,0.0113091115,0.0055861,0.005541444,0.6724969,0.012675701,0.28128648],"study_design_scores_gemma":[0.000051579107,0.00021745895,0.0038297493,0.0054210997,0.00018171176,0.0026397991,0.0051436634,0.022137374,0.003814149,0.36466014,0.5916977,0.00020553442],"about_ca_topic_score_codex":0.008182196,"about_ca_topic_score_gemma":0.009818329,"teacher_disagreement_score":0.017833598,"about_ca_system_score_codex":0.0050313324,"about_ca_system_score_gemma":0.0135206,"threshold_uncertainty_score":0.07291365},"labels":[],"label_agreement":null},{"id":"W2796115009","doi":"10.1007/s10664-018-9617-6","title":"Studying the consistency of star ratings and reviews of popular free hybrid Android and iOS apps","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Mobile and Web Applications","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"JavaScript; Android (operating system); Computer science; Codebase; Mobile apps; World Wide Web; App store; Web application; Source code; Operating system","score_opus":0.02501936444025835,"score_gpt":0.2575023618221422,"score_spread":0.23248299738188385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2796115009","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9922523,0.0008443329,0.0011237889,0.00026734784,0.00006928344,0.00006606244,0.0012382646,0.000039556217,0.004099025],"genre_scores_gemma":[0.99605834,0.00019963621,0.0007864315,0.00007413944,0.000057481026,0.000058985177,0.0018277306,0.000028237637,0.0009091269],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9789898,0.007393023,0.003044313,0.0021852723,0.007888806,0.00049877685],"domain_scores_gemma":[0.51830304,0.29896548,0.0879975,0.014312307,0.07650794,0.003913716],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018894937,0.00026216582,0.00053607963,0.0077699306,0.00068011344,0.0024599147,0.00083011546,0.0008778798,0.0018616202],"category_scores_gemma":[0.24052672,0.00029229044,0.0006488347,0.005515909,0.00076885836,0.0026469901,0.0014336024,0.0008779391,0.0007179856],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008678882,0.00014810565,0.96177906,0.0004420311,0.0007599005,0.00009695385,0.0058536385,0.00044025364,0.0014950298,0.00051175524,0.003018821,0.024586601],"study_design_scores_gemma":[0.000029032628,0.00030031695,0.98978525,0.000078068486,0.00018160594,0.0001560915,0.0028532818,0.0024159595,0.0006604812,0.0002469504,0.0032432904,0.000049768616],"about_ca_topic_score_codex":0.0033547794,"about_ca_topic_score_gemma":0.005488254,"teacher_disagreement_score":0.018894937,"about_ca_system_score_codex":0.0008268688,"about_ca_system_score_gemma":0.0005909089,"threshold_uncertainty_score":0.09992719},"labels":[],"label_agreement":null},{"id":"W2796164377","doi":"10.1007/s10664-018-9615-8","title":"An empirical study of Android Wear user complaints","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Wearable computer; Android (operating system); Wearable technology; Internet privacy; Mobile device; Computer science; Categorization; Empirical research; Complaint; World Wide Web; Human–computer interaction; Artificial intelligence","score_opus":0.020782005223002522,"score_gpt":0.3194921780268243,"score_spread":0.2987101728038218,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2796164377","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9987245,0.000053825577,0.000097211814,0.000065513996,0.0000034064649,0.000021064652,0.00008962792,0.000006359236,0.00093843794],"genre_scores_gemma":[0.9988599,0.00007129173,0.00015909453,0.00004738318,0.0000105013705,0.000022559498,0.00015492404,0.0000053218605,0.000669096],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9979249,0.0008690514,0.00018009146,0.00017687584,0.00062204385,0.00022691995],"domain_scores_gemma":[0.89335215,0.076957874,0.016346296,0.0029670114,0.0084066875,0.0019699936],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020736668,0.0003078323,0.00025221967,0.0021640018,0.00092548365,0.0013844051,0.00060449186,0.00091873575,0.0026454534],"category_scores_gemma":[0.039195422,0.0003168692,0.00027280563,0.0016378094,0.0009117203,0.0017817622,0.0008936428,0.0014242837,0.0007942114],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027483978,0.002332556,0.9558785,0.00021834079,0.000047175487,0.00064295036,0.016856816,0.00017928651,0.0011490626,0.00062315044,0.0012060432,0.020591263],"study_design_scores_gemma":[0.000021256612,0.0011108418,0.97082806,0.00008634482,0.000050015413,0.00097625,0.022432566,0.0016320096,0.0008589545,0.00016133585,0.0018185689,0.00002379808],"about_ca_topic_score_codex":0.0038153927,"about_ca_topic_score_gemma":0.0054165656,"teacher_disagreement_score":0.0038153927,"about_ca_system_score_codex":0.0006074772,"about_ca_system_score_gemma":0.0008343017,"threshold_uncertainty_score":0.010966718},"labels":[],"label_agreement":null},{"id":"W2801496046","doi":"10.1007/s10664-018-9614-9","title":"Investigating whether and how software developers understand open source software licensing","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Systems, Applications & Products in Data Processing (Canada); University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"License; Context (archaeology); Open source software; Computer science; World Wide Web; Open source; Software; Source code; Software engineering; Software development; Data science; Knowledge management","score_opus":0.04796779462994141,"score_gpt":0.28552701820068405,"score_spread":0.23755922357074263,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2801496046","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9897377,0.000106259,0.0015001959,0.00080896815,0.000006572278,0.000013928302,0.000025008356,0.000009312681,0.0077920044],"genre_scores_gemma":[0.99777335,0.0001077911,0.00043461117,0.00019427674,0.000005447856,0.000009244289,0.00003701191,0.000013161772,0.0014250221],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.99556535,0.0019079054,0.00022790105,0.0004323235,0.0013351489,0.0005313386],"domain_scores_gemma":[0.85106117,0.113513924,0.018231614,0.00455795,0.010302186,0.002333033],"candidate_categories":["metaresearch","open_science"],"consensus_categories":[],"category_scores_codex":[0.0066358084,0.00018525824,0.00018415645,0.0010160388,0.0008964395,0.0036067308,0.00066811027,0.0016830548,0.0047353506],"category_scores_gemma":[0.12175564,0.0003183575,0.0001879298,0.0011006582,0.001829036,0.009350756,0.0021016493,0.0018386338,0.0006566622],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019623655,0.0009995583,0.8174478,0.00017568198,0.00007819609,0.00028710783,0.09965776,0.00082139444,0.0039074505,0.016374001,0.002134409,0.05792052],"study_design_scores_gemma":[0.00007366587,0.00028643227,0.79709864,0.00031799186,0.00010135705,0.00034482943,0.13925275,0.008255553,0.004079621,0.030984756,0.019128053,0.00007630654],"about_ca_topic_score_codex":0.011322269,"about_ca_topic_score_gemma":0.021785475,"teacher_disagreement_score":0.9993319,"about_ca_system_score_codex":0.0015858299,"about_ca_system_score_gemma":0.0026773252,"threshold_uncertainty_score":0.035093963},"labels":[],"label_agreement":null},{"id":"W2805828232","doi":"10.1007/s10664-018-9629-2","title":"What can Android mobile app developers do about the energy consumption of machine learning?","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Green IT and Sustainability","field":"Engineering","cited_by":49,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Alberta","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Machine learning; Artificial intelligence; Android (operating system); Energy consumption; Implementation; Facial recognition system; Multimedia; Feature extraction; Software engineering; Operating system; Engineering","score_opus":0.010641704860067385,"score_gpt":0.23646973270715824,"score_spread":0.22582802784709086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805828232","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8911701,0.0016942015,0.008374098,0.04271619,0.00026504122,0.000046871017,0.0005892372,0.000102813356,0.055041518],"genre_scores_gemma":[0.9934163,0.0006633966,0.0012155866,0.0018550467,0.00006408559,0.00001913088,0.00015814754,0.000060176433,0.002548106],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9975873,0.0010019047,0.00009720694,0.00026733248,0.00078859547,0.0002577048],"domain_scores_gemma":[0.9509992,0.03133533,0.006080543,0.002110457,0.008179818,0.0012946302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047295587,0.00031275416,0.00025031035,0.0011037478,0.00068296713,0.003527635,0.000602431,0.0016313773,0.0037344957],"category_scores_gemma":[0.067748465,0.0003036527,0.00030579933,0.0010685236,0.0017583126,0.006404267,0.000836166,0.002357901,0.0006914005],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049832324,0.0003971002,0.6769636,0.00045749216,0.0002838811,0.0006182529,0.040998805,0.0026599744,0.0032437153,0.027854439,0.020094752,0.22592972],"study_design_scores_gemma":[0.00004415457,0.00028899105,0.76127875,0.0011774396,0.00024295882,0.0005978967,0.11027607,0.009834565,0.0042398158,0.04983604,0.06200388,0.00017951104],"about_ca_topic_score_codex":0.012951261,"about_ca_topic_score_gemma":0.018275803,"teacher_disagreement_score":0.012951261,"about_ca_system_score_codex":0.0012692411,"about_ca_system_score_gemma":0.0013371896,"threshold_uncertainty_score":0.02575177},"labels":[],"label_agreement":null},{"id":"W2810627707","doi":"10.1007/s10664-018-9634-5","title":"How do developers utilize source code from stack overflow?","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":129,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Source code; Copying; Codebase; Code reuse; Reuse; Code review; Programming language; Code (set theory); Stack (abstract data type); Software engineering; World Wide Web; Operating system; Static program analysis; Software; Software development; Engineering; Set (abstract data type)","score_opus":0.03258012208955109,"score_gpt":0.27706487837659804,"score_spread":0.24448475628704697,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2810627707","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9576204,0.0008502946,0.022813613,0.0040575853,0.00009546603,0.0002550305,0.0005785972,0.0022729319,0.011456121],"genre_scores_gemma":[0.96653,0.001053503,0.019524798,0.0011638462,0.000058496113,0.0002727035,0.0015550312,0.0022516674,0.007589886],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9830078,0.005370624,0.0012919918,0.0019796395,0.0071321046,0.0012178249],"domain_scores_gemma":[0.8591958,0.08143572,0.020903405,0.013463951,0.022463614,0.0025374452],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.016612967,0.00074211456,0.00044804285,0.00627989,0.0023224752,0.004476807,0.0014784412,0.0017224196,0.00217407],"category_scores_gemma":[0.17544097,0.0010115751,0.00061381405,0.0040213913,0.0028559975,0.012467347,0.0042458773,0.001496183,0.0014599632],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034679435,0.00034267074,0.28949273,0.0010266607,0.00011724637,0.003472979,0.36116168,0.0009053434,0.010014689,0.005389312,0.015231742,0.31249812],"study_design_scores_gemma":[0.00011519407,0.0008215495,0.47661754,0.0030917965,0.0002780044,0.0069891536,0.21481434,0.00983843,0.021569867,0.012170052,0.25308362,0.0006104334],"about_ca_topic_score_codex":0.010385393,"about_ca_topic_score_gemma":0.011947373,"teacher_disagreement_score":0.98338705,"about_ca_system_score_codex":0.002656428,"about_ca_system_score_gemma":0.004246478,"threshold_uncertainty_score":0.087858796},"labels":[],"label_agreement":null},{"id":"W2883022143","doi":"10.1007/s10664-018-9640-7","title":"GreenScaler: training software energy models with automatic test generation","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Green IT and Sustainability","field":"Engineering","cited_by":57,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Energy consumption; Software; Computer science; Heuristics; Heuristic; Energy (signal processing); Reliability engineering; Embedded system; Software engineering; Engineering; Artificial intelligence; Operating system","score_opus":0.027419903582851166,"score_gpt":0.22244643836317063,"score_spread":0.19502653478031945,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2883022143","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17304356,0.00050250185,0.5430091,0.0007265401,0.00038048776,0.00038148894,0.00512099,0.26969197,0.007143395],"genre_scores_gemma":[0.63807565,0.00018092956,0.34002206,0.0005097256,0.000047741,0.0006077134,0.009938106,0.006154786,0.0044631856],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99924916,0.0002908861,0.00005315364,0.00019651071,0.00015290339,0.00005734003],"domain_scores_gemma":[0.99435425,0.004051445,0.00018625053,0.00085484714,0.00043514374,0.000118081334],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014695416,0.0016852467,0.0005060408,0.0009302655,0.00023481723,0.00057307206,0.0029221894,0.0014190476,0.010894062],"category_scores_gemma":[0.0119341295,0.0010234046,0.0007515688,0.000583801,0.00043574476,0.0019104141,0.0011408465,0.0019221305,0.0035704107],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000848715,0.0010308578,0.0091001615,0.0004659891,0.000252216,0.00026854913,0.00021601176,0.4462598,0.00729432,0.0022801044,0.058223467,0.47375986],"study_design_scores_gemma":[0.00010135046,0.000080513484,0.0004769593,0.000017951525,0.000021412403,0.000029860388,0.000016378253,0.9900478,0.005138732,0.0017603348,0.002297748,0.000010930381],"about_ca_topic_score_codex":0.006351678,"about_ca_topic_score_gemma":0.009412625,"teacher_disagreement_score":0.010894062,"about_ca_system_score_codex":0.0006471296,"about_ca_system_score_gemma":0.0010105955,"threshold_uncertainty_score":0.036444247},"labels":[],"label_agreement":null},{"id":"W2886347592","doi":"10.1007/s10664-018-9636-3","title":"An empirical study on the issue reports with questions raised during the issue resolving process","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Beihang University","keywords":"Computer science; Process (computing); Data science; Eclipse; Empirical research","score_opus":0.020315733376175562,"score_gpt":0.3259201812352616,"score_spread":0.3056044478590861,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2886347592","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99540436,0.00015465186,0.0013047103,0.00018165063,0.00002406068,0.0002180242,0.00014011528,0.000021034388,0.0025513628],"genre_scores_gemma":[0.9942584,0.0002095655,0.0026597378,0.00020020522,0.000058208563,0.00027718156,0.0004603054,0.000031715248,0.0018446964],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9689794,0.022005502,0.0022598724,0.0014708978,0.004544881,0.0007393977],"domain_scores_gemma":[0.40338665,0.4972746,0.05679219,0.016714202,0.021221,0.0046113343],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.021783276,0.0005484095,0.00046256144,0.0031251279,0.0021017182,0.0034583104,0.0015079395,0.001776516,0.004710596],"category_scores_gemma":[0.3017368,0.00063842064,0.00050533947,0.0034554426,0.0019474891,0.0035363801,0.0023301325,0.003128713,0.0013211583],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032394512,0.025179578,0.70758355,0.0012213668,0.00022906666,0.0021650908,0.12995586,0.0010297414,0.007862128,0.0046789423,0.0036278402,0.113227464],"study_design_scores_gemma":[0.00041423336,0.0080002025,0.8304944,0.00052365306,0.0003618708,0.0027380919,0.114890926,0.0059742974,0.012439423,0.0025858632,0.021338888,0.00023806165],"about_ca_topic_score_codex":0.0021330796,"about_ca_topic_score_gemma":0.0020130398,"teacher_disagreement_score":0.9782167,"about_ca_system_score_codex":0.0011731879,"about_ca_system_score_gemma":0.0017594008,"threshold_uncertainty_score":0.11520237},"labels":[],"label_agreement":null},{"id":"W2888049099","doi":"10.1007/s10664-018-9643-4","title":"Preventing duplicate bug reports by continuously querying bug reports","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Mitacs; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Data deduplication; Software bug; BitTorrent tracker; Search engine indexing; Security bug; Information retrieval; Software; Database; Artificial intelligence; Cloud computing; Programming language","score_opus":0.014794330250701704,"score_gpt":0.2740173627620474,"score_spread":0.2592230325113457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2888049099","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5870996,0.006114238,0.34987333,0.0026452632,0.0007105131,0.0009826222,0.0037443158,0.042711478,0.0061185197],"genre_scores_gemma":[0.7615736,0.000798263,0.22943056,0.00045228578,0.00027922523,0.00015996677,0.0044342997,0.0008494577,0.002022242],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97736037,0.004419712,0.0022367102,0.004793025,0.0103961695,0.0007939152],"domain_scores_gemma":[0.85971427,0.06763268,0.026074817,0.023316788,0.020497924,0.0027635216],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009665112,0.002384076,0.0025754205,0.010113718,0.0010165529,0.003865131,0.0047631366,0.0033334896,0.0013642854],"category_scores_gemma":[0.10799349,0.0012404303,0.0010119183,0.005081452,0.0007849063,0.005985355,0.0035241086,0.002490349,0.0014929232],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00128154,0.0015134009,0.20277809,0.0011366296,0.0005844301,0.0006830449,0.0014251282,0.014181882,0.048715666,0.0021386854,0.018297106,0.7072645],"study_design_scores_gemma":[0.0005775551,0.0027840785,0.13184156,0.0004928637,0.0019281843,0.0047625476,0.0021578725,0.73313504,0.08256289,0.01753617,0.021802014,0.00041916926],"about_ca_topic_score_codex":0.0056443163,"about_ca_topic_score_gemma":0.006150316,"teacher_disagreement_score":0.010113718,"about_ca_system_score_codex":0.0007718369,"about_ca_system_score_gemma":0.003785508,"threshold_uncertainty_score":0.05111462},"labels":[],"label_agreement":null},{"id":"W2897662303","doi":"10.1007/s10664-018-9656-z","title":"High-level software requirements and iteration changes: a predictive model","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Systems, Applications & Products in Data Processing (Canada); University of Victoria","funders":"","keywords":"Computer science; Software engineering; Reliability engineering; Engineering","score_opus":0.05761095084229382,"score_gpt":0.3005239212894088,"score_spread":0.242912970447115,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2897662303","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90352726,0.00042181058,0.08411842,0.0012217442,0.00006325852,0.00013591466,0.0006967203,0.00069194974,0.009123077],"genre_scores_gemma":[0.9949923,0.00011828767,0.0030490926,0.000039847782,0.000016824055,0.000045920802,0.00027955853,0.000040037405,0.0014181362],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985355,0.00058683095,0.00006205454,0.00032132625,0.00030904062,0.0001852555],"domain_scores_gemma":[0.95819336,0.03492821,0.0019644476,0.002219778,0.0020431874,0.0006510645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00424396,0.0012235928,0.00092181476,0.002590059,0.00068312534,0.003018691,0.0029327006,0.002491886,0.0057912865],"category_scores_gemma":[0.034580328,0.0010906435,0.0015381845,0.0020302914,0.0014615301,0.0041576945,0.0012201297,0.002728245,0.0016983738],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001097263,0.0018962864,0.18506393,0.00019719142,0.00038595725,0.000593857,0.0013549557,0.7164422,0.0011167426,0.022722725,0.0034497255,0.06567913],"study_design_scores_gemma":[0.000032055148,0.00009868672,0.01517061,0.00002330983,0.00008614695,0.000090132155,0.00010276857,0.97434366,0.00020056892,0.009611786,0.00021905027,0.00002129125],"about_ca_topic_score_codex":0.017453218,"about_ca_topic_score_gemma":0.011113468,"teacher_disagreement_score":0.017453218,"about_ca_system_score_codex":0.0016054567,"about_ca_system_score_gemma":0.0016623931,"threshold_uncertainty_score":0.034703255},"labels":[],"label_agreement":null},{"id":"W2899817180","doi":"10.1007/s10664-018-9665-y","title":"An empirical study of patch uplift in rapid release development pipelines","year":2018,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Geology; Channel (broadcasting); Lead (geology); Computer science; Paleontology; Telecommunications","score_opus":0.03303537718028712,"score_gpt":0.3249213934817224,"score_spread":0.2918860163014353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2899817180","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9977724,0.000050621275,0.0003737557,0.00013475225,0.0000024638614,0.000022076743,0.000055244018,0.000021239743,0.0015673697],"genre_scores_gemma":[0.999032,0.00003213973,0.0002881403,0.000024260595,0.0000049496925,0.000012470908,0.00010276831,0.000008322919,0.00049496646],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9972378,0.0008858359,0.00019940197,0.00039559798,0.0008620489,0.0004193709],"domain_scores_gemma":[0.8134998,0.12250689,0.04277676,0.0068884986,0.00921692,0.00511119],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004777736,0.00024571503,0.0002572842,0.0017466173,0.0010938628,0.0016039403,0.0010264306,0.0011097135,0.0055405325],"category_scores_gemma":[0.096568696,0.0003822416,0.0002652953,0.0019686662,0.0012756481,0.0037048075,0.0013754087,0.0022367162,0.000756643],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007566218,0.0028353329,0.9227969,0.00016113442,0.00005829839,0.0007578267,0.012045006,0.0028155833,0.0019787946,0.0047385385,0.0025731812,0.048482828],"study_design_scores_gemma":[0.00007935046,0.0012288487,0.9656322,0.00008407593,0.000049186587,0.00034078956,0.014136304,0.011784343,0.00095718406,0.0018108425,0.0038609812,0.000035946232],"about_ca_topic_score_codex":0.008194193,"about_ca_topic_score_gemma":0.009634121,"teacher_disagreement_score":0.008194193,"about_ca_system_score_codex":0.0013950785,"about_ca_system_score_gemma":0.0013529642,"threshold_uncertainty_score":0.025267363},"labels":[],"label_agreement":null},{"id":"W2907872047","doi":"10.1007/s10664-018-9677-7","title":"An empirical study of DLL injection bugs in the Firefox ecosystem","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Polytechnique Montréal","keywords":"Software bug; Software; Host (biology); Computer science; Operating system; Biology","score_opus":0.013934126707237858,"score_gpt":0.29050260713577125,"score_spread":0.2765684804285334,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2907872047","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9990331,0.00004566919,0.0002368813,0.00009543963,0.0000018037381,0.000015937369,0.00006274677,0.000009178895,0.0004992733],"genre_scores_gemma":[0.9987382,0.00005344121,0.00057394477,0.00005264676,0.000005483057,0.000016436634,0.00015158081,0.0000072550506,0.00040093105],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99832886,0.0007878477,0.00011481053,0.00019256433,0.00036802594,0.00020787836],"domain_scores_gemma":[0.91295433,0.05975143,0.01571322,0.0032266027,0.006571857,0.0017825484],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0037106446,0.00036006636,0.00019675175,0.0025045758,0.0009329626,0.0008138744,0.0007131939,0.0009514971,0.0016139968],"category_scores_gemma":[0.039940935,0.00028278097,0.0002038555,0.0015403272,0.0013147158,0.0022943027,0.0008849123,0.0014869106,0.0004475321],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028562668,0.003989591,0.95702255,0.000103710365,0.000052252894,0.0006367831,0.006096903,0.0007302192,0.0016561502,0.00175872,0.0012260585,0.026441433],"study_design_scores_gemma":[0.00007796608,0.0011387989,0.9674781,0.00011667451,0.00006222889,0.0014490872,0.010033155,0.014200995,0.0017545188,0.0012654111,0.0023837343,0.00003949825],"about_ca_topic_score_codex":0.0077777724,"about_ca_topic_score_gemma":0.012132171,"teacher_disagreement_score":0.9962894,"about_ca_system_score_codex":0.0010232662,"about_ca_system_score_gemma":0.00083396304,"threshold_uncertainty_score":0.019623995},"labels":[],"label_agreement":null},{"id":"W2911789761","doi":"10.1007/s10664-018-9671-0","title":"Automatic query reformulation for code search using crowdsourced knowledge","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Computer science; Information retrieval; Web search query; Java; Code (set theory); Query expansion; Search engine; Web query classification; XPath; Precision and recall; Natural language; Programming language; World Wide Web; Artificial intelligence; XML; Set (abstract data type)","score_opus":0.0489429157531969,"score_gpt":0.33474163569030263,"score_spread":0.2857987199371057,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2911789761","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08637204,0.0021826508,0.8493398,0.002100178,0.00050653814,0.0017377636,0.009045261,0.034911204,0.013804503],"genre_scores_gemma":[0.42132717,0.0007229435,0.5393496,0.0006939675,0.00028367955,0.0009966008,0.025087005,0.002061641,0.009477247],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99297935,0.0018670734,0.00051925133,0.0012200715,0.0028828748,0.00053133897],"domain_scores_gemma":[0.98790026,0.005119511,0.00051581755,0.002669304,0.0034288252,0.00036633792],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025995432,0.0014978731,0.0018692572,0.007863757,0.0021715374,0.0026772642,0.002772435,0.002239003,0.014339402],"category_scores_gemma":[0.023804503,0.0005852267,0.0016973112,0.004772567,0.001071148,0.004800716,0.005934762,0.0016621961,0.006777961],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010769182,0.00079614477,0.0033213915,0.0017116144,0.00019550504,0.0007083854,0.0022844686,0.018321684,0.05665681,0.016914438,0.11044601,0.7875666],"study_design_scores_gemma":[0.0003655837,0.0004220522,0.004428033,0.0003410807,0.00031658972,0.0007783648,0.0038087082,0.7850179,0.06859254,0.054328077,0.081371166,0.00022982674],"about_ca_topic_score_codex":0.014811992,"about_ca_topic_score_gemma":0.020216096,"teacher_disagreement_score":0.014811992,"about_ca_system_score_codex":0.0018420956,"about_ca_system_score_gemma":0.004259887,"threshold_uncertainty_score":0.047970116},"labels":[],"label_agreement":null},{"id":"W2914035201","doi":"10.1007/s10664-019-09684-y","title":"Towards prioritizing user-related issue reports of mobile applications","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Prioritization; Computer science; Android (operating system); World Wide Web; Empirical research; Mobile apps; Process (computing); Matching (statistics); Data science; Internet privacy; Engineering; Process management","score_opus":0.00985313260491852,"score_gpt":0.2752562160281317,"score_spread":0.26540308342321317,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914035201","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6087324,0.005301878,0.34316728,0.00396585,0.0006752288,0.0024356106,0.0060716546,0.0068027093,0.022847261],"genre_scores_gemma":[0.7348778,0.0012635555,0.2522573,0.00034182565,0.0003490186,0.0003812512,0.005472346,0.00036714113,0.0046897647],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98289675,0.0054452815,0.0026429805,0.0013962132,0.006555807,0.0010629455],"domain_scores_gemma":[0.8228139,0.08433201,0.02644898,0.006084991,0.055697855,0.0046221744],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018070482,0.0017968118,0.0010930492,0.025633285,0.001437125,0.009897017,0.0017515486,0.0016781728,0.0026060964],"category_scores_gemma":[0.11961482,0.00068792165,0.0009395056,0.007549374,0.00047147562,0.005174757,0.0023112493,0.0016938291,0.0018675535],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001025759,0.00080135395,0.32162836,0.001772918,0.0002627872,0.00052878197,0.0030372099,0.0040267026,0.03455808,0.0054752086,0.0101180095,0.6167649],"study_design_scores_gemma":[0.00024173694,0.0022032103,0.50598246,0.0018967872,0.0015444782,0.0023143368,0.021367233,0.25235182,0.116292775,0.029470779,0.065905444,0.00042895178],"about_ca_topic_score_codex":0.008769181,"about_ca_topic_score_gemma":0.016055726,"teacher_disagreement_score":0.025633285,"about_ca_system_score_codex":0.001313951,"about_ca_system_score_gemma":0.0057641147,"threshold_uncertainty_score":0.09556693},"labels":[],"label_agreement":null},{"id":"W2914582798","doi":"10.1007/s10664-018-9679-5","title":"The impact of feature reduction techniques on defect prediction models","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Industrial Vision Systems and Defect Detection","field":"Engineering","cited_by":99,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Alberta","funders":"Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada","keywords":"Reduction (mathematics); Feature (linguistics); Computer science; Pattern recognition (psychology); Artificial intelligence; Data mining; Mathematics","score_opus":0.015841601350863765,"score_gpt":0.25694013528353327,"score_spread":0.2410985339326695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914582798","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60608673,0.005455708,0.37620327,0.0018300486,0.00028821913,0.000108747496,0.0011583416,0.004365603,0.0045033577],"genre_scores_gemma":[0.9222826,0.0007645126,0.07395417,0.00013227928,0.00011764433,0.00003202606,0.0011835222,0.0001611523,0.0013720747],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998353,0.00077576004,0.000116203475,0.0002999745,0.0003336922,0.00012146447],"domain_scores_gemma":[0.9639595,0.031620283,0.0007300295,0.001886629,0.0016580295,0.0001454389],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036136194,0.0012027622,0.0011450988,0.0014307725,0.00037093606,0.0011887618,0.0010328577,0.0009811593,0.0016543594],"category_scores_gemma":[0.02470349,0.0003781466,0.0011378293,0.001068913,0.00034244865,0.002086393,0.00047368702,0.0015855423,0.0005932565],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010828036,0.0008157186,0.016510697,0.0001654875,0.00035410954,0.00010359835,0.000053384134,0.3927362,0.0055345986,0.0012878652,0.002996939,0.5783587],"study_design_scores_gemma":[0.000019259542,0.00013952325,0.002836507,0.000010515222,0.00007050117,0.00004054817,0.000016682072,0.9936801,0.0018047655,0.0011492275,0.00022371045,0.000008596446],"about_ca_topic_score_codex":0.010634134,"about_ca_topic_score_gemma":0.007165811,"teacher_disagreement_score":0.010634134,"about_ca_system_score_codex":0.00042540216,"about_ca_system_score_gemma":0.0010042761,"threshold_uncertainty_score":0.02114445},"labels":[],"label_agreement":null},{"id":"W2920526824","doi":"10.1007/s10664-019-09695-9","title":"An empirical study of the long duration of continuous integration builds","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":82,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Duration (music); Computer science; Set (abstract data type); Empirical research; Software development; Cache; Software; Software engineering; Process management; Engineering; Statistics; Operating system; Programming language","score_opus":0.016944779324193943,"score_gpt":0.2973705753052279,"score_spread":0.280425795981034,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2920526824","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9938267,0.00044293577,0.0009305709,0.00027331413,0.000010512202,0.000014942018,0.00012132221,0.000018588977,0.004360995],"genre_scores_gemma":[0.99891543,0.00008873181,0.00024330943,0.000029043515,0.000013790086,0.000009839599,0.000110230394,0.000007832455,0.0005818794],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9978466,0.0007039348,0.00015201532,0.00027117654,0.00077824836,0.00024791397],"domain_scores_gemma":[0.82757074,0.1255152,0.027043065,0.0063669286,0.00743384,0.0060701943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052403007,0.00019149082,0.00026113354,0.0013831147,0.0011793369,0.0025686766,0.0010296764,0.0010469686,0.0071107224],"category_scores_gemma":[0.07605561,0.00027619922,0.00020758006,0.0023573313,0.0012128501,0.0034088078,0.0018285424,0.0020077599,0.00073262275],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014391749,0.0016558745,0.8718597,0.00026344808,0.00013646408,0.0007902127,0.014317352,0.0016224755,0.002770205,0.01442274,0.0014666948,0.089255475],"study_design_scores_gemma":[0.000091879956,0.0015740215,0.9559712,0.00013733881,0.00015538324,0.00075946125,0.016751532,0.0053416993,0.0015385112,0.009628538,0.007986527,0.00006396296],"about_ca_topic_score_codex":0.0030757294,"about_ca_topic_score_gemma":0.0048115104,"teacher_disagreement_score":0.0071107224,"about_ca_system_score_codex":0.001044201,"about_ca_system_score_gemma":0.0012643309,"threshold_uncertainty_score":0.027713716},"labels":[],"label_agreement":null},{"id":"W2920929439","doi":"10.1007/s10664-019-09690-0","title":"Extracting and studying the Logging-Code-Issue- Introducing changes in Java-based large-scale open source software systems","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Logging; Debugging; Java; Source code; Correctness; Code (set theory); Software; Code review; Database; Software engineering; Static program analysis; Software development; Programming language","score_opus":0.018345802225703055,"score_gpt":0.26551265940598784,"score_spread":0.24716685718028478,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2920929439","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99267334,0.00012219373,0.005954134,0.000056128676,0.000014352554,0.000029513463,0.00036675323,0.00025287468,0.00053073216],"genre_scores_gemma":[0.992841,0.000081718485,0.0057020197,0.000016174543,0.000012907221,0.000017824641,0.000875867,0.000078633704,0.00037391763],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.997993,0.0002764604,0.00023137117,0.0005036036,0.0008504169,0.00014517648],"domain_scores_gemma":[0.9607843,0.019911481,0.009205152,0.0032955192,0.0060877753,0.0007157883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015372772,0.00025696697,0.00022387673,0.0030831073,0.00040782068,0.0008600295,0.00053774443,0.00041351077,0.00051528437],"category_scores_gemma":[0.026016654,0.00025373607,0.00034980755,0.002451094,0.0004258462,0.0015996522,0.0006582613,0.0008398867,0.00019696668],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015311643,0.000415799,0.8352122,0.00033034943,0.00008573946,0.00042005468,0.002069513,0.0037765575,0.014348536,0.0009874637,0.0011982181,0.14100245],"study_design_scores_gemma":[0.000009413896,0.00010769052,0.9443297,0.00003807294,0.00006669206,0.00031468496,0.0008125183,0.043092955,0.008274079,0.0010463607,0.0018809335,0.000026877067],"about_ca_topic_score_codex":0.0046126842,"about_ca_topic_score_gemma":0.009563486,"teacher_disagreement_score":0.0046126842,"about_ca_system_score_codex":0.00052019313,"about_ca_system_score_gemma":0.0007670968,"threshold_uncertainty_score":0.009171665},"labels":[],"label_agreement":null},{"id":"W2921594963","doi":"10.1007/s10664-019-09691-z","title":"Assessing and optimizing the performance impact of the just-in-time configuration parameters - a case study on PyPy","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Reliability engineering; Engineering","score_opus":0.03857829876151698,"score_gpt":0.31762281544977644,"score_spread":0.27904451668825947,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2921594963","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99033254,0.000105015744,0.007749113,0.00007367966,0.000006365576,0.000051290426,0.00007270601,0.00041957092,0.0011896025],"genre_scores_gemma":[0.98853403,0.00005097857,0.010692519,0.000014520705,0.0000027658568,0.000021437894,0.00009173746,0.000094246345,0.00049781514],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964065,0.0018414428,0.0001988652,0.00045284693,0.00076498894,0.00033539],"domain_scores_gemma":[0.94614005,0.040937524,0.0026143226,0.0066106003,0.003008332,0.00068918435],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044340324,0.0008703152,0.00050631317,0.0008845444,0.0005957846,0.0011196812,0.0018189315,0.0010827967,0.0014548635],"category_scores_gemma":[0.043176048,0.0003947959,0.0002868169,0.0011255222,0.00102032,0.0024033377,0.0009521174,0.0013077498,0.00030351346],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005317246,0.010456096,0.11693261,0.0012788558,0.00025709296,0.0016785525,0.0035455308,0.35591057,0.09464117,0.007138381,0.0030082264,0.39983565],"study_design_scores_gemma":[0.00046716176,0.008538543,0.111352555,0.0001523769,0.00029095498,0.0009997693,0.0024909815,0.7400948,0.12426668,0.0055607394,0.005591042,0.0001944356],"about_ca_topic_score_codex":0.0032454943,"about_ca_topic_score_gemma":0.0028493756,"teacher_disagreement_score":0.0044340324,"about_ca_system_score_codex":0.0008463407,"about_ca_system_score_gemma":0.0012143563,"threshold_uncertainty_score":0.023449719},"labels":[],"label_agreement":null},{"id":"W2935734788","doi":"10.1007/s10664-019-09703-y","title":"iPerfDetector: Characterizing and detecting performance anti-patterns in iOS applications","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Green IT and Sustainability","field":"Engineering","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Workload; Mobile apps; Open source; Thread (computing); Empirical research; Performance improvement; Operating system; Software; World Wide Web; Software engineering; Engineering","score_opus":0.006439130811232374,"score_gpt":0.20538353318306649,"score_spread":0.1989444023718341,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2935734788","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.93650264,0.0004419971,0.03810786,0.00026025396,0.00006485614,0.000103424085,0.005188296,0.013863847,0.0054668863],"genre_scores_gemma":[0.96567017,0.00013298733,0.027691955,0.000090894944,0.00003178716,0.00008835832,0.0033461626,0.0003788245,0.0025688577],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99912184,0.00012314055,0.00007409654,0.00021387612,0.00035694512,0.00011004267],"domain_scores_gemma":[0.9929055,0.0031753115,0.0018074028,0.0009691506,0.000919619,0.00022307839],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009701923,0.0007038038,0.0003043004,0.0035299489,0.0002620576,0.00077775755,0.0007292802,0.00048745528,0.0016448285],"category_scores_gemma":[0.008750983,0.00021852148,0.00028136585,0.0019642753,0.00032883277,0.0011273592,0.00066666456,0.00058572856,0.00074776355],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078969804,0.0007756294,0.5381277,0.00044833534,0.00018745332,0.00038846157,0.0006226037,0.008355611,0.03440951,0.0022623923,0.014475359,0.39915726],"study_design_scores_gemma":[0.0001160819,0.0012941341,0.49677864,0.000119573975,0.00017503006,0.0014127356,0.0006248371,0.4131868,0.06328025,0.0053380793,0.017563222,0.000110682915],"about_ca_topic_score_codex":0.0032218532,"about_ca_topic_score_gemma":0.0047473125,"teacher_disagreement_score":0.0035299489,"about_ca_system_score_codex":0.00035348663,"about_ca_system_score_gemma":0.0006024488,"threshold_uncertainty_score":0.006406188},"labels":[],"label_agreement":null},{"id":"W2941522337","doi":"10.1007/s10664-019-09711-y","title":"Characterizing industry-academia collaborations in software engineering: evidence from 101 projects","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":62,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"ITEA3; Estonian Research Competency Council; University of Calgary; Tartu Ülikool; Eesti Teadusagentuur; Tekes; Universidade do Minho; Norges Forskningsråd; ITEA; Strategic Research Council; Fundação para a Ciência e a Tecnologia; Åbo Akademi; Universidade Federal de Santa Catarina","keywords":"Context (archaeology); Relevance (law); Engineering; Empirical research; Engineering management; Business; Knowledge management; Marketing; Computer science; Political science; Geography","score_opus":0.046992848767424926,"score_gpt":0.2945148710774645,"score_spread":0.24752202231003956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2941522337","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"incentives","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"incentives","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9799177,0.007504795,0.0027357915,0.0017836556,0.00004532897,0.0003830328,0.0021290765,0.00003498218,0.0054656817],"genre_scores_gemma":[0.9855025,0.0050745886,0.0046135522,0.0004689167,0.000043698794,0.0007480847,0.0029824271,0.00004112779,0.00052509696],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9038764,0.054301724,0.011540457,0.0057477932,0.021883614,0.0026499724],"domain_scores_gemma":[0.51886606,0.32521322,0.07600653,0.017225701,0.052674953,0.01001352],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.062412266,0.0004832568,0.00066534424,0.01263432,0.0025869824,0.0044597127,0.0016784774,0.0019893423,0.002178815],"category_scores_gemma":[0.23381253,0.00070396706,0.000818677,0.018660175,0.0023103873,0.006585874,0.0068231863,0.0014843026,0.0007804694],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004014808,0.0012160552,0.7184652,0.009837482,0.0005393252,0.0013804281,0.088991545,0.0007082244,0.00084868603,0.0037921665,0.008286961,0.16553238],"study_design_scores_gemma":[0.00019324306,0.0010554614,0.7805858,0.009976752,0.00046745516,0.001278164,0.15010649,0.001755063,0.0011714398,0.0027038534,0.050556805,0.0001494566],"about_ca_topic_score_codex":0.005252487,"about_ca_topic_score_gemma":0.00982689,"teacher_disagreement_score":0.93758774,"about_ca_system_score_codex":0.0026494307,"about_ca_system_score_gemma":0.0070435763,"threshold_uncertainty_score":0.3300715},"labels":[],"label_agreement":null},{"id":"W2944738881","doi":"10.1007/s10664-019-09704-x","title":"cregit: Token-level blame information in git version control repositories","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Victoria","funders":"","keywords":"Blame; Commit; Security token; Computer science; Source code; Code (set theory); Computer security; Programming language; Psychology; Social psychology; Set (abstract data type); Database","score_opus":0.012995523476119583,"score_gpt":0.24247411488059042,"score_spread":0.22947859140447086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2944738881","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05015705,0.0005438632,0.12287758,0.0013330362,0.00037306378,0.00036618562,0.068323456,0.7428213,0.0132044945],"genre_scores_gemma":[0.48187178,0.0005806695,0.1477177,0.00081672583,0.00033913885,0.0005057821,0.27904165,0.0750179,0.014108642],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.99364716,0.0011562118,0.0008388964,0.0010623957,0.0027053033,0.00059000036],"domain_scores_gemma":[0.9535688,0.012430215,0.003786698,0.024117386,0.004576163,0.0015206257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0069120456,0.001490458,0.0009179097,0.007960364,0.0010047528,0.0048786066,0.002893015,0.0023790663,0.016584583],"category_scores_gemma":[0.05846504,0.0015566888,0.001055984,0.006626611,0.0010192649,0.011594651,0.0036205226,0.003045109,0.011029558],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0046914606,0.00086189475,0.05192275,0.0016590894,0.00039590008,0.0009510065,0.0024485497,0.016227618,0.012713255,0.02725656,0.47053355,0.41033834],"study_design_scores_gemma":[0.0015723577,0.0011725771,0.05338402,0.0011503906,0.0006094202,0.0023219,0.0008562885,0.26788014,0.1395916,0.08166092,0.44846636,0.0013340526],"about_ca_topic_score_codex":0.0048859804,"about_ca_topic_score_gemma":0.005285226,"teacher_disagreement_score":0.016584583,"about_ca_system_score_codex":0.0015367526,"about_ca_system_score_gemma":0.0025112615,"threshold_uncertainty_score":0.055480957},"labels":[],"label_agreement":null},{"id":"W2945530585","doi":"10.1007/s10664-019-09719-4","title":"Fostering good coding practices through individual feedback and gamification: an industrial case study","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Coding (social sciences); Variety (cybernetics); Computer science; Best practice; Code review; Software; Quality (philosophy); Software engineering; Knowledge management; Software quality; Data science; Software development; Management","score_opus":0.1720305715459125,"score_gpt":0.37070826398116774,"score_spread":0.19867769243525524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945530585","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9905956,0.000037964255,0.00637731,0.00021076004,0.0000072896655,0.00018220498,0.000013608462,0.000059437163,0.002515754],"genre_scores_gemma":[0.9833059,0.000051763964,0.015321982,0.00003996227,0.0000036339643,0.000099608325,0.000021282884,0.000016286778,0.0011396612],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9950151,0.0031465637,0.0001589203,0.00036743036,0.00077013735,0.00054178486],"domain_scores_gemma":[0.9541145,0.033474654,0.0023590932,0.0046997676,0.002958063,0.0023939894],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.010050518,0.0007212798,0.00037374144,0.0013913481,0.0021428291,0.0019307119,0.0019714208,0.0018792866,0.0017233766],"category_scores_gemma":[0.03139336,0.00040394146,0.00032623482,0.0009321265,0.002145197,0.001470084,0.0026925707,0.0015960177,0.00036390754],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003174185,0.0573215,0.13368413,0.00090333406,0.00016036138,0.0063621863,0.10651066,0.032266267,0.022857388,0.016454607,0.004168423,0.616137],"study_design_scores_gemma":[0.0032610483,0.052707862,0.17532249,0.001693288,0.0006325358,0.010524913,0.18989782,0.3676045,0.10715235,0.04738656,0.042923257,0.00089339],"about_ca_topic_score_codex":0.002017097,"about_ca_topic_score_gemma":0.0044750436,"teacher_disagreement_score":0.98994946,"about_ca_system_score_codex":0.0014228699,"about_ca_system_score_gemma":0.0023665843,"threshold_uncertainty_score":0.05315286},"labels":[],"label_agreement":null},{"id":"W2945826489","doi":"10.1007/s10664-019-09788-5","title":"MSRBot: Using bots to answer questions from software repositories","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Software; Process (computing); Software development; Work (physics); Code (set theory); Software analytics; Team software process; Source code; Software peer review; Verification and validation","score_opus":0.04104006454653197,"score_gpt":0.29623689857983276,"score_spread":0.2551968340333008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945826489","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42823046,0.0015868532,0.39150453,0.0052291206,0.0010672666,0.002322056,0.017983435,0.12403758,0.028038627],"genre_scores_gemma":[0.6853954,0.000481928,0.27351126,0.0021110484,0.00021204336,0.0014351932,0.017362762,0.0025535745,0.016936846],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9946234,0.0025964368,0.0003350834,0.0007430606,0.0014365925,0.00026548133],"domain_scores_gemma":[0.9725859,0.020389713,0.0015356172,0.0028049336,0.001826438,0.0008573316],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040606447,0.001219173,0.00079019764,0.003911489,0.0011750847,0.0014091885,0.0014836356,0.0024289144,0.0056175217],"category_scores_gemma":[0.03231308,0.0005253039,0.0005248169,0.0014579347,0.0007297025,0.004845444,0.003626405,0.0014198284,0.00413536],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030012142,0.002534155,0.08138831,0.0039001373,0.00058106886,0.0019320701,0.014815794,0.010289581,0.04333743,0.020229934,0.18298322,0.635007],"study_design_scores_gemma":[0.00079879776,0.0023220426,0.0508546,0.00082436454,0.00038585675,0.001536536,0.012039081,0.5880781,0.0373174,0.10580501,0.19965313,0.0003851582],"about_ca_topic_score_codex":0.0048720436,"about_ca_topic_score_gemma":0.010518932,"teacher_disagreement_score":0.0056175217,"about_ca_system_score_codex":0.00087518786,"about_ca_system_score_gemma":0.0013829935,"threshold_uncertainty_score":0.021474957},"labels":[],"label_agreement":null},{"id":"W2946357416","doi":"10.1007/s10664-019-09700-1","title":"An empirical study on the teams structures in social coding using GitHub projects","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Code review; Computer science; Team software process; Coding (social sciences); Project team; Software development; Process (computing); Software engineering; Software; Social network (sociolinguistics); Knowledge management; Software development process; World Wide Web; Software quality; Social media","score_opus":0.04951311858025997,"score_gpt":0.33549298622300366,"score_spread":0.2859798676427437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946357416","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9973967,0.00002732358,0.0009025713,0.00006322919,0.000002605908,0.000033173055,0.000036606885,0.000006919706,0.001530818],"genre_scores_gemma":[0.99849355,0.000026563475,0.0008516102,0.000014272853,0.000003624135,0.000046338213,0.000094168594,0.000012357852,0.00045755706],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98943955,0.007307339,0.0003279047,0.00064474,0.001456688,0.00082382344],"domain_scores_gemma":[0.86578107,0.09962747,0.0146505935,0.0066321455,0.008992263,0.004316373],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.007586372,0.00035027074,0.0002794229,0.003940901,0.002504173,0.002569251,0.0012176532,0.0010012363,0.0022488667],"category_scores_gemma":[0.08155632,0.000312679,0.00022550556,0.0049597574,0.003213228,0.0037766655,0.0035122025,0.001529144,0.00038585404],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004891693,0.001427693,0.72194403,0.00022009281,0.000054339598,0.00048894354,0.19689535,0.0010546788,0.0019754437,0.009933477,0.0013389173,0.06417778],"study_design_scores_gemma":[0.00005949001,0.00044317092,0.7258902,0.00018388037,0.000045178956,0.0003644382,0.25104868,0.008818034,0.0016318895,0.0063303807,0.0051211603,0.00006343099],"about_ca_topic_score_codex":0.015316621,"about_ca_topic_score_gemma":0.0223064,"teacher_disagreement_score":0.99241364,"about_ca_system_score_codex":0.002630683,"about_ca_system_score_gemma":0.002863227,"threshold_uncertainty_score":0.04012108},"labels":[],"label_agreement":null},{"id":"W2946929989","doi":"10.1007/s10664-020-09853-4","title":"Automating system test case classification and prioritization for use case-driven testing in product lines","year":2020,"lang":"en","type":"preprint","venue":"Empirical Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"H2020 European Research Council; Fonds National de la Recherche Luxembourg","keywords":"Computer science; Requirement prioritization; Test strategy; Context (archaeology); Domain engineering; Test case; Software product line; New product development; Risk-based testing; Software engineering; Software development; Systems engineering; Reliability engineering; Software; Engineering; Machine learning; Requirement; Component-based software engineering; Software construction","score_opus":0.18065782686550533,"score_gpt":0.34379375934561823,"score_spread":0.1631359324801129,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946929989","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31254593,0.00048080456,0.6693227,0.0004009847,0.000058369165,0.0009650152,0.0008863943,0.012572959,0.0027668101],"genre_scores_gemma":[0.56460726,0.0001178908,0.431242,0.00010025405,0.00002215597,0.00029767465,0.0017582153,0.00040338744,0.0014512222],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99360037,0.002350032,0.00057872426,0.0008774572,0.0020456424,0.0005477193],"domain_scores_gemma":[0.966861,0.021450873,0.0028569342,0.0027094092,0.0051949336,0.0009268684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004927715,0.0013360209,0.0012652256,0.005162145,0.0006778929,0.002556637,0.0018827072,0.0010744046,0.0031084837],"category_scores_gemma":[0.027094468,0.00070007634,0.00081219675,0.0017488153,0.00048057726,0.0013403485,0.0015654975,0.0012660489,0.0012733769],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006098682,0.0010830197,0.05314092,0.00044273204,0.00013108544,0.00032879566,0.0007567275,0.032660134,0.05312477,0.0020710912,0.0062137987,0.84943706],"study_design_scores_gemma":[0.00013867959,0.00035074144,0.033958923,0.00011470723,0.00011677313,0.00051625917,0.00046350135,0.9012247,0.05299494,0.006314626,0.0037415859,0.0000645781],"about_ca_topic_score_codex":0.009783866,"about_ca_topic_score_gemma":0.016622148,"teacher_disagreement_score":0.009783866,"about_ca_system_score_codex":0.00096120656,"about_ca_system_score_gemma":0.0029370426,"threshold_uncertainty_score":0.026060522},"labels":[],"label_agreement":null},{"id":"W2950236608","doi":"10.1007/s10664-019-09709-6","title":"A study of build inflation in 30 million CPAN builds on 13 Perl versions and 10 operating systems","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Polytechnique Montréal","funders":"","keywords":"Perl; Inflation (cosmology); Computer science; Operating system; Programming language; Physics; Astronomy","score_opus":0.021393192071352304,"score_gpt":0.28106298312999833,"score_spread":0.259669791058646,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950236608","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9988696,0.000037575635,0.00009544048,0.00007244752,0.0000022320237,0.000003141552,0.0002163593,0.000011188925,0.0006919959],"genre_scores_gemma":[0.99917245,0.000017626264,0.000058733935,0.0000144414935,0.0000051852267,0.0000029322105,0.00044471247,0.0000054639927,0.00027830937],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9985839,0.0004068725,0.00008286235,0.0002061501,0.00046967826,0.00025049003],"domain_scores_gemma":[0.97516114,0.013136293,0.006395646,0.0008895278,0.0031963123,0.0012210915],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.001766184,0.00023069514,0.00026981026,0.0013098102,0.00063455256,0.0011018997,0.0006418735,0.00063866226,0.0016205243],"category_scores_gemma":[0.015758624,0.00037020585,0.00036734328,0.002509547,0.00056436076,0.0011835583,0.0006348196,0.0015616645,0.00041736092],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088094064,0.00063267845,0.9737879,0.000029638975,0.00013457352,0.00037784342,0.0012564866,0.0059433845,0.000739736,0.0012961759,0.0020594287,0.012861221],"study_design_scores_gemma":[0.0000138450605,0.00023605287,0.9840206,0.000007914457,0.000041366093,0.00014764628,0.0014397546,0.012426332,0.00047182105,0.00034879535,0.0008244439,0.000021483747],"about_ca_topic_score_codex":0.029236505,"about_ca_topic_score_gemma":0.02771707,"teacher_disagreement_score":0.9982338,"about_ca_system_score_codex":0.0019864342,"about_ca_system_score_gemma":0.0007186261,"threshold_uncertainty_score":0.05813265},"labels":[],"label_agreement":null},{"id":"W2954653546","doi":"10.1007/s10664-019-09718-5","title":"The inconsistency between theory and practice in managing inconsistency in requirements engineering","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Israel Science Foundation; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Consistency (knowledge bases); Perception; Identification (biology); Field (mathematics); Sight; Phenomenon; Psychology; Computer science; Management science; Knowledge management; Epistemology; Engineering; Artificial intelligence","score_opus":0.018747001289031427,"score_gpt":0.2946853165192847,"score_spread":0.27593831523025325,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2954653546","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16995728,0.011650854,0.65820843,0.1227883,0.00114383,0.00045381262,0.00012260159,0.00048563912,0.035189163],"genre_scores_gemma":[0.7831143,0.0018866719,0.20716359,0.0056768353,0.0004158418,0.00039473802,0.000106224055,0.00019594544,0.0010458374],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.6795399,0.2065517,0.02316469,0.011476529,0.075823314,0.0034438025],"domain_scores_gemma":[0.21924037,0.68574643,0.025164237,0.03039973,0.037718467,0.0017307275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.18606207,0.00073013135,0.0014101752,0.007924855,0.0036548646,0.012056914,0.004961023,0.0072785714,0.0018388805],"category_scores_gemma":[0.5884252,0.0016818494,0.0010382156,0.005108254,0.015378339,0.03446011,0.009082831,0.010850004,0.00036583212],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040661695,0.0005943454,0.02835573,0.0025249023,0.00050320604,0.00071370305,0.033976424,0.01547237,0.0019417026,0.46269956,0.0060182214,0.44679323],"study_design_scores_gemma":[0.00020652293,0.00045268162,0.0111727575,0.004081302,0.00034584466,0.0009602857,0.0172303,0.052322045,0.0038206433,0.8902728,0.018878354,0.00025661415],"about_ca_topic_score_codex":0.0037515427,"about_ca_topic_score_gemma":0.003759853,"teacher_disagreement_score":0.18606207,"about_ca_system_score_codex":0.008169004,"about_ca_system_score_gemma":0.01304503,"threshold_uncertainty_score":0.98400205},"labels":[],"label_agreement":null},{"id":"W2955854775","doi":"10.1007/s10664-019-09733-6","title":"Identifying gameplay videos that exhibit bugs in computer games","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Digital Games and Media","field":"Social Sciences","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Queen's University","funders":"","keywords":"Computer science; Metadata; Context (archaeology); Game Developer; Quality (philosophy); Classifier (UML); Multimedia; World Wide Web; Information retrieval; Human–computer interaction; Artificial intelligence; Game design","score_opus":0.0285645952500537,"score_gpt":0.2942501658445961,"score_spread":0.2656855705945424,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2955854775","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99794966,0.0000642444,0.0008455812,0.00005442911,0.000011592518,0.00006458521,0.0002186357,0.000042324642,0.0007490326],"genre_scores_gemma":[0.9971723,0.000054183514,0.0016447834,0.000042831758,0.000005111486,0.00003890537,0.00050363765,0.00002066962,0.0005176338],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9986274,0.00047050932,0.0001253857,0.00021509932,0.00037734048,0.0001843077],"domain_scores_gemma":[0.9667182,0.021523638,0.0068449974,0.001051607,0.0027539562,0.0011076666],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011252816,0.00052910513,0.00023106828,0.0031059713,0.0005740974,0.00096876617,0.00064233557,0.0009718543,0.0018018942],"category_scores_gemma":[0.037728213,0.00032229113,0.00023533768,0.0010138568,0.00047953415,0.0011112029,0.00087308313,0.00079266145,0.00028864358],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079184474,0.0021312486,0.9230985,0.0002941624,0.0001439903,0.00086336164,0.0038745024,0.00085599895,0.0072106062,0.0009159484,0.0023690085,0.057450946],"study_design_scores_gemma":[0.00008626836,0.0013270675,0.96459585,0.00021907527,0.00017175257,0.002415067,0.006929859,0.015120599,0.004982019,0.0011044214,0.0029947911,0.000053198855],"about_ca_topic_score_codex":0.007549031,"about_ca_topic_score_gemma":0.017541356,"teacher_disagreement_score":0.007549031,"about_ca_system_score_codex":0.0006102406,"about_ca_system_score_gemma":0.0006308901,"threshold_uncertainty_score":0.015010178},"labels":[],"label_agreement":null},{"id":"W2963079908","doi":"10.1007/s10664-019-09743-4","title":"CAPS: a supervised technique for classifying Stack Overflow posts concerning API issues","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan; Queen's University","funders":"","keywords":"Computer science; Documentation; Application programming interface; Task (project management); Conditional random field; Readability; Field (mathematics); Baseline (sea); Data science; World Wide Web; Artificial intelligence; Machine learning; Engineering; Programming language","score_opus":0.0362815258937799,"score_gpt":0.3089535097084091,"score_spread":0.2726719838146292,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963079908","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37205988,0.0023793096,0.51544863,0.00083468785,0.0009740865,0.0024741306,0.02841619,0.0635778,0.013835296],"genre_scores_gemma":[0.4863953,0.0005604911,0.44510603,0.00041949027,0.0008418864,0.0013729015,0.046711177,0.0014534275,0.017139254],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9970414,0.0003988564,0.00037483155,0.00076101994,0.0011434715,0.0002802962],"domain_scores_gemma":[0.98970175,0.0037004002,0.0016152938,0.0013434131,0.003022974,0.000616258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022646796,0.0018874274,0.0011332909,0.010824432,0.0016282775,0.0015019494,0.0022615993,0.0020687603,0.0044256058],"category_scores_gemma":[0.0071877185,0.00044744337,0.0013070058,0.004916966,0.0007287445,0.002806585,0.001840433,0.0020537945,0.004101608],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009274927,0.0015421641,0.034193315,0.0009683235,0.00036460266,0.0006143465,0.00077339576,0.0049295845,0.04866533,0.0024416877,0.079598986,0.82498074],"study_design_scores_gemma":[0.00026146672,0.0011871892,0.06604637,0.00030956068,0.0006210195,0.0017550472,0.0018770604,0.7732438,0.087597266,0.010152664,0.056666207,0.00028233897],"about_ca_topic_score_codex":0.00590141,"about_ca_topic_score_gemma":0.01536159,"teacher_disagreement_score":0.010824432,"about_ca_system_score_codex":0.00066802814,"about_ca_system_score_gemma":0.0031385904,"threshold_uncertainty_score":0.014805138},"labels":[],"label_agreement":null},{"id":"W2966112319","doi":"10.1007/s10664-019-09744-3","title":"Bounties on technical Q&amp;A sites: a case study of Stack Overflow bounties","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Expert finding and Q&A systems","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Queen's University","funders":"","keywords":"Reputation; Stack (abstract data type); Computer science; Data science; Operations research; World Wide Web; Engineering; Sociology; Social science","score_opus":0.036897091978036166,"score_gpt":0.29205392285738857,"score_spread":0.2551568308793524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2966112319","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9937552,0.00008446132,0.0007959978,0.00046444315,0.000010824676,0.00007098925,0.00005816856,0.00006064087,0.0046992213],"genre_scores_gemma":[0.99512863,0.00010407326,0.0014186254,0.00012595327,0.000011030189,0.000028195758,0.000084397885,0.000040493578,0.0030586557],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9961171,0.001439275,0.00017865535,0.00037211893,0.00085822504,0.0010346539],"domain_scores_gemma":[0.95900375,0.025121965,0.004820978,0.0023970495,0.0031368874,0.005519424],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046970993,0.0005653956,0.00038798104,0.0033213277,0.010808574,0.0034044303,0.0024336374,0.004076649,0.0068340967],"category_scores_gemma":[0.022961885,0.0007746092,0.0005006989,0.003361035,0.004164733,0.0046694423,0.003886248,0.0034627537,0.0008212165],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018633136,0.007832038,0.46372175,0.00062639336,0.00030096612,0.06482823,0.2215822,0.011934538,0.0068383394,0.030595295,0.018654121,0.17122278],"study_design_scores_gemma":[0.00037849927,0.003563165,0.39103788,0.0011349863,0.00033047487,0.01409366,0.45537436,0.04307867,0.0067868563,0.0129904095,0.0708499,0.00038112642],"about_ca_topic_score_codex":0.050483204,"about_ca_topic_score_gemma":0.08291581,"teacher_disagreement_score":0.050483204,"about_ca_system_score_codex":0.00434225,"about_ca_system_score_gemma":0.0056727957,"threshold_uncertainty_score":0.10037869},"labels":[],"label_agreement":null},{"id":"W2967780468","doi":"10.1007/s10664-019-09736-3","title":"The impact of context metrics on just-in-time defect prediction","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada","keywords":"Source lines of code; Context (archaeology); Metric (unit); Computer science; Measure (data warehouse); Machine learning; Software metric; Software; Data mining; Artificial intelligence; Statistics; Software quality; Mathematics; Software development; Programming language; Engineering; Operations management","score_opus":0.020471050761877507,"score_gpt":0.29267956045619137,"score_spread":0.27220850969431387,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2967780468","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9762323,0.0017151366,0.017995195,0.00050771236,0.0001448427,0.000033031472,0.0010686578,0.0005726783,0.0017303476],"genre_scores_gemma":[0.9952567,0.00009406863,0.003887393,0.00002570634,0.000034071483,0.000007530691,0.0005017814,0.000033190234,0.00015949813],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9933216,0.0031915302,0.0005431055,0.0012094405,0.0013324588,0.0004017906],"domain_scores_gemma":[0.83823,0.12933287,0.009538211,0.010793479,0.009162536,0.0029429155],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0073770573,0.0011225974,0.00084540615,0.002664894,0.00048980577,0.0017082756,0.00086500996,0.0011719442,0.0010298378],"category_scores_gemma":[0.0904149,0.00021368168,0.0005335115,0.0020630793,0.0005178119,0.004036544,0.0011877618,0.0014589666,0.0003470936],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013092734,0.0007091391,0.79001445,0.00020198133,0.00034526055,0.00018838403,0.00021994098,0.03706287,0.002314791,0.0012821052,0.0027647384,0.16358714],"study_design_scores_gemma":[0.000086557455,0.002515583,0.32797652,0.00013142591,0.00040948132,0.0006979722,0.00053259614,0.6522227,0.00506515,0.008336649,0.0019243264,0.00010103265],"about_ca_topic_score_codex":0.003641223,"about_ca_topic_score_gemma":0.0072049634,"teacher_disagreement_score":0.0073770573,"about_ca_system_score_codex":0.00049118756,"about_ca_system_score_gemma":0.00096017716,"threshold_uncertainty_score":0.0390141},"labels":[],"label_agreement":null},{"id":"W2969694848","doi":"10.1007/s10664-019-09760-3","title":"Empirical study of android repackaged applications","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi; Concordia University","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Malware; Android (operating system); Computer science; Popularity; Mobile malware; Upload; Download; Computer security; World Wide Web; Empirical research; Publication; Internet privacy; Operating system; Advertising","score_opus":0.014366851444132947,"score_gpt":0.28824754942819064,"score_spread":0.2738806979840577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2969694848","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9987269,0.00008602659,0.00018243938,0.0000594034,0.0000027202084,0.000014733931,0.00008084342,0.000010629722,0.00083630334],"genre_scores_gemma":[0.99873847,0.00008142591,0.00023387959,0.000023009721,0.0000069889297,0.000010504521,0.00017676479,0.0000064709934,0.00072250416],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.998831,0.00040552203,0.00009007159,0.00014981105,0.00041036386,0.000113415306],"domain_scores_gemma":[0.9554478,0.030317578,0.0063241036,0.0024955545,0.004741967,0.00067311025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016259421,0.00036108936,0.00019030205,0.0014511434,0.00047520114,0.0008092583,0.0006451879,0.0005985008,0.0021599506],"category_scores_gemma":[0.027476382,0.00019995341,0.00022907188,0.0010454287,0.0006902372,0.0015353259,0.0005423486,0.0011824842,0.00070318364],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075812906,0.0037644587,0.9018617,0.00045593173,0.00012032019,0.0014718478,0.006756319,0.0019914391,0.006332555,0.003910559,0.0029126722,0.06966412],"study_design_scores_gemma":[0.000052264375,0.0014433955,0.95632046,0.0001533963,0.00012382503,0.0021596758,0.007490423,0.02048066,0.0056669395,0.0011090552,0.0049574315,0.000042452728],"about_ca_topic_score_codex":0.0027890666,"about_ca_topic_score_gemma":0.0033550346,"teacher_disagreement_score":0.0027890666,"about_ca_system_score_codex":0.00040785785,"about_ca_system_score_gemma":0.0004660091,"threshold_uncertainty_score":0.008598924},"labels":[],"label_agreement":null},{"id":"W2971660263","doi":"10.1007/s10664-019-09771-0","title":"Why reinventing the wheels? An empirical study on library reuse and re-implementation","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Digital and Traditional Archives Management","field":"Arts and Humanities","cited_by":47,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reuse; Empirical research; Computer science; Engineering; Engineering management; Mathematics; Statistics; Waste management","score_opus":0.044619323073125015,"score_gpt":0.2730376624820939,"score_spread":0.22841833940896886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2971660263","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9859913,0.0004632718,0.0015349482,0.0021842432,0.000015159039,0.000087368295,0.000022602202,0.000034470522,0.009666728],"genre_scores_gemma":[0.99598134,0.00022169312,0.00095139886,0.00026869573,0.0000063214625,0.00003579256,0.000023089344,0.00004171171,0.0024698963],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.96410894,0.019700317,0.002295594,0.0017173816,0.008223014,0.003954661],"domain_scores_gemma":[0.73433083,0.18110527,0.031330574,0.0208286,0.0234225,0.0089821825],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.032697104,0.00033163128,0.00076699007,0.006570247,0.00797303,0.011712064,0.0032933692,0.0029788695,0.0043053813],"category_scores_gemma":[0.1720726,0.0008527535,0.00071373954,0.009169232,0.010816117,0.015903667,0.008901084,0.0049052658,0.00072834635],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024819194,0.001979373,0.22381523,0.0004768283,0.00008880051,0.0009581066,0.58577025,0.0005220285,0.0008769019,0.036376692,0.0028075583,0.14608005],"study_design_scores_gemma":[0.00004531439,0.0003438644,0.1559732,0.0007730216,0.00011501616,0.00041093255,0.8015041,0.0021562271,0.0011972918,0.007875224,0.029502967,0.00010285233],"about_ca_topic_score_codex":0.034477692,"about_ca_topic_score_gemma":0.05004683,"teacher_disagreement_score":0.034477692,"about_ca_system_score_codex":0.007777967,"about_ca_system_score_gemma":0.02238711,"threshold_uncertainty_score":0.17292082},"labels":[],"label_agreement":null},{"id":"W2983752450","doi":"10.1007/s10664-019-09784-9","title":"An experimental scrutiny of visual design modelling: VCL up against UML+OCL","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Data Visualization and Analytics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Unified Modeling Language; Visual modeling; Comprehension; Notation; Software engineering; Programming language; Software; Mathematics","score_opus":0.04332768921336913,"score_gpt":0.3190907181283654,"score_spread":0.27576302891499627,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2983752450","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9898594,0.000076432494,0.005363362,0.00015386798,0.00004438634,0.00056273764,0.00006449379,0.00008325647,0.003792089],"genre_scores_gemma":[0.9875223,0.000054815217,0.009214248,0.000107374166,0.000034557714,0.001439149,0.00010726926,0.00009249076,0.0014277712],"study_design_codex":"qualitative","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.97526044,0.017200064,0.0015494267,0.0025638498,0.0027480463,0.0006782675],"domain_scores_gemma":[0.6597048,0.30183008,0.012139272,0.016196365,0.0085171675,0.0016122686],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020859709,0.0008294821,0.0010171391,0.0011096188,0.0014030373,0.0035689962,0.0016552032,0.001739938,0.008040471],"category_scores_gemma":[0.18097755,0.00072285224,0.00070342084,0.0009934367,0.004433906,0.0033364282,0.0044473046,0.0025403583,0.0009520988],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.024667064,0.040345784,0.032235835,0.0047024405,0.00036041057,0.0014878539,0.41784132,0.01585133,0.1794612,0.03828235,0.005107139,0.23965731],"study_design_scores_gemma":[0.011556802,0.14719371,0.18817064,0.0028248855,0.0012192309,0.0016017176,0.1448295,0.08817034,0.25504017,0.06604195,0.092018604,0.0013324125],"about_ca_topic_score_codex":0.0009323481,"about_ca_topic_score_gemma":0.00060514064,"teacher_disagreement_score":0.020859709,"about_ca_system_score_codex":0.0020839383,"about_ca_system_score_gemma":0.00068064325,"threshold_uncertainty_score":0.110317945},"labels":[],"label_agreement":null},{"id":"W2988589793","doi":"10.1007/s10664-019-09776-9","title":"Guest Editorial: Special Issue on Software Engineering for Mobile Applications","year":2019,"lang":"en","type":"editorial","venue":"Empirical Software Engineering","topic":"Green IT and Sustainability","field":"Engineering","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Software engineering; Systems engineering; Engineering","score_opus":0.005986996702665983,"score_gpt":0.25187963848650285,"score_spread":0.24589264178383688,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2988589793","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00002194738,0.003045096,0.00009625558,0.023751346,0.9715182,0.000014470545,0.000051384945,0.000031025178,0.001470211],"genre_scores_gemma":[0.00020191006,0.0019037928,0.00007211028,0.008271601,0.98072696,0.000016457738,0.000032255863,0.00004108484,0.008733732],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99218774,0.0013188014,0.00084564503,0.0009775212,0.004183241,0.00048703112],"domain_scores_gemma":[0.9601553,0.01557352,0.003120963,0.0010482464,0.014838039,0.0052639213],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0098631885,0.0045193266,0.004384882,0.0059700916,0.0038217306,0.012418339,0.0032136603,0.018225152,0.033559125],"category_scores_gemma":[0.03781627,0.0013180904,0.0027113948,0.0022377232,0.0026880032,0.004659142,0.0023483827,0.017272351,0.018660689],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022122038,0.0000109711045,0.000018111028,0.00013004137,0.000011949934,0.00006433845,0.000006023411,0.00001661622,0.00003645537,0.00017271869,0.9969289,0.0025818408],"study_design_scores_gemma":[0.00008228465,0.000032696847,0.00034632892,0.0005537007,0.00006427664,0.00017969277,0.000044085402,0.00021793232,0.00010670027,0.0016600032,0.9966905,0.000021693315],"about_ca_topic_score_codex":0.0018854404,"about_ca_topic_score_gemma":0.00680271,"teacher_disagreement_score":0.033559125,"about_ca_system_score_codex":0.0036097977,"about_ca_system_score_gemma":0.004091772,"threshold_uncertainty_score":0.11226642},"labels":[],"label_agreement":null},{"id":"W3000116190","doi":"10.1007/s10664-019-09790-x","title":"What should your run-time configuration framework do to help developers?","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Université du Québec à Chicoutimi; Queen's University","funders":"","keywords":"Software engineering; Configuration Management (ITSM); Computer science; Software configuration management; Software deployment; Debugging; Domain (mathematical analysis); Flexibility (engineering); Requirements engineering; Systems engineering; Code refactoring; Software system; Software; Engineering; Software construction; Operating system","score_opus":0.103734205417861,"score_gpt":0.33745975405508927,"score_spread":0.23372554863722828,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3000116190","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32362145,0.006151846,0.21143235,0.32795718,0.002841469,0.00064970984,0.0011307088,0.025264982,0.1009504],"genre_scores_gemma":[0.8584744,0.0016652674,0.11916065,0.00850281,0.00041223777,0.00021427996,0.00050204486,0.0016991834,0.009369176],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9879071,0.006267544,0.00050044374,0.0010210611,0.0030864528,0.0012174365],"domain_scores_gemma":[0.91121244,0.035456985,0.009243378,0.0141862165,0.019086856,0.010814209],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02835548,0.0012252715,0.0007322273,0.0021669257,0.0018734345,0.0063228877,0.0029703209,0.004332868,0.009513755],"category_scores_gemma":[0.13011105,0.00078434637,0.00047798915,0.0015729391,0.0020145788,0.020085862,0.0015752623,0.0031308641,0.005461243],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041505747,0.0017916516,0.11042027,0.0007202568,0.00020158887,0.00041461515,0.005099489,0.0026993162,0.0073604346,0.025221664,0.1209654,0.7246903],"study_design_scores_gemma":[0.0013019365,0.003435256,0.3198184,0.005886114,0.00074246794,0.005196494,0.04561276,0.06334477,0.028534908,0.17817798,0.3467116,0.0012372966],"about_ca_topic_score_codex":0.005221488,"about_ca_topic_score_gemma":0.008907817,"teacher_disagreement_score":0.9716445,"about_ca_system_score_codex":0.001732095,"about_ca_system_score_gemma":0.00739406,"threshold_uncertainty_score":0.14995992},"labels":[],"label_agreement":null},{"id":"W3001783472","doi":"10.1007/s10664-020-09807-w","title":"Ammonia: an approach for deriving project-specific bug patterns","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Java; Software bug; Software regression; Security bug; Software development; Source code; Limiting; Pointer (user interface); Software maintenance","score_opus":0.0743664019156529,"score_gpt":0.30118913813212417,"score_spread":0.22682273621647125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3001783472","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037677683,0.00011981761,0.9351523,0.00027787662,0.00006537372,0.0007223539,0.002360571,0.021061191,0.0025628302],"genre_scores_gemma":[0.099550895,0.00010875428,0.89341533,0.000091975024,0.000027611064,0.00067192764,0.0032858134,0.0010729098,0.0017747233],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99536514,0.0009905199,0.0006477621,0.0011302839,0.0016611929,0.0002050526],"domain_scores_gemma":[0.98513675,0.0055111265,0.0027858615,0.0027986409,0.0033766003,0.0003911018],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038281824,0.0014707645,0.00087465433,0.008736753,0.0009987787,0.0023339852,0.0017978718,0.0012194605,0.002756398],"category_scores_gemma":[0.025865888,0.0010206068,0.0014094635,0.00497124,0.00066346175,0.0024532252,0.0024485502,0.0012561668,0.0014528795],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003287719,0.00049540255,0.10563283,0.0013833855,0.00037945432,0.0014754318,0.004136307,0.021953043,0.033638775,0.014658751,0.016673513,0.7992443],"study_design_scores_gemma":[0.00017010723,0.0006991554,0.049265824,0.00067390443,0.0005906918,0.0033378527,0.0022599152,0.793428,0.045848876,0.03036886,0.073033385,0.0003234141],"about_ca_topic_score_codex":0.0044375323,"about_ca_topic_score_gemma":0.006944282,"teacher_disagreement_score":0.008736753,"about_ca_system_score_codex":0.00073885446,"about_ca_system_score_gemma":0.0031267728,"threshold_uncertainty_score":0.020245612},"labels":[],"label_agreement":null},{"id":"W3004570974","doi":"10.1007/s10664-019-09781-y","title":"How bugs are born: a model to identify how bugs are introduced in software components","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":74,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; University of Waterloo","funders":"H2020 Industrial Leadership; Ministerio de Asuntos Económicos y Transformación Digital, Gobierno de España; Nederlandse Organisatie voor Wetenschappelijk Onderzoek","keywords":"Software bug; Software regression; Computer science; False positive paradox; Source lines of code; Software; Source code; Open source; Debugging; Software maintenance; Snapshot (computer storage); Code (set theory); Security bug; Data mining; Software development; Software quality; Programming language; Machine learning; Database; Operating system; Set (abstract data type); Software security assurance","score_opus":0.062116288198003355,"score_gpt":0.2983208692382943,"score_spread":0.23620458104029096,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3004570974","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.62440807,0.0021242509,0.35925683,0.0036663774,0.00010529376,0.00035073224,0.0028883952,0.003922355,0.003277577],"genre_scores_gemma":[0.9424195,0.00027681788,0.05380355,0.0002613464,0.000043990505,0.00011864904,0.0021675325,0.0001172587,0.00079130486],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958449,0.0012613453,0.00038031166,0.001313354,0.0008557821,0.00034424895],"domain_scores_gemma":[0.93993104,0.04215767,0.008307773,0.0034703333,0.004912148,0.001221157],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070214965,0.0013356216,0.0010450677,0.00929131,0.0009392814,0.0038458325,0.0021303005,0.0030915784,0.001811786],"category_scores_gemma":[0.049292944,0.0007353934,0.0018044963,0.003447786,0.0025210644,0.006107893,0.0023212172,0.0016079808,0.0007701086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013689456,0.0005830828,0.5505127,0.001018617,0.00052371534,0.0011827888,0.0034109924,0.26920176,0.006679441,0.023908623,0.012215618,0.12939379],"study_design_scores_gemma":[0.000047697282,0.00014182208,0.033474814,0.0000883963,0.00007424544,0.00042687793,0.00045079426,0.9415368,0.0012370471,0.020761326,0.0017055571,0.000054610784],"about_ca_topic_score_codex":0.011138844,"about_ca_topic_score_gemma":0.009765143,"teacher_disagreement_score":0.011138844,"about_ca_system_score_codex":0.0023772153,"about_ca_system_score_gemma":0.0015975679,"threshold_uncertainty_score":0.037133694},"labels":[],"label_agreement":null},{"id":"W3011992028","doi":"10.1007/s10664-019-09796-5","title":"An exploratory study of smart contracts in the Ethereum blockchain platform","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":241,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Queen's University","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Microsoft","keywords":"Blockchain; Smart contract; Computer science; Computer security","score_opus":0.03360663077197832,"score_gpt":0.26758092421985136,"score_spread":0.23397429344787304,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3011992028","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99322927,0.000033346765,0.0010371694,0.000199928,0.00000202559,0.000055337325,0.00004351441,0.0000047880167,0.005394596],"genre_scores_gemma":[0.9975224,0.000037993843,0.0006480396,0.000025975307,0.0000022891397,0.000025195275,0.00005363338,0.0000042649135,0.0016801353],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.99841845,0.00094414235,0.000045432076,0.000121928904,0.00026753274,0.00020244045],"domain_scores_gemma":[0.97211117,0.023128057,0.0016244892,0.0010537518,0.0010249572,0.0010575498],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040579573,0.00019185797,0.00023775018,0.0008804653,0.0023137783,0.0016318237,0.0008919979,0.0013313796,0.0071343263],"category_scores_gemma":[0.018267484,0.00022244861,0.00012586472,0.0016244611,0.0022897155,0.004655004,0.0015409823,0.0017056372,0.00051315513],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019864368,0.013374854,0.29528242,0.00060129556,0.000086053566,0.007224159,0.15204938,0.014923802,0.010870099,0.39138567,0.0044796555,0.10773615],"study_design_scores_gemma":[0.00040718005,0.00430611,0.30363452,0.0005129401,0.00006493028,0.0017347657,0.40004608,0.117541894,0.009334632,0.11933172,0.042930372,0.00015475534],"about_ca_topic_score_codex":0.0047358633,"about_ca_topic_score_gemma":0.0071626026,"teacher_disagreement_score":0.0071343263,"about_ca_system_score_codex":0.0015292328,"about_ca_system_score_gemma":0.0016992342,"threshold_uncertainty_score":0.023866653},"labels":[],"label_agreement":null},{"id":"W3013163251","doi":"10.1007/s10664-019-09783-w","title":"Building the perfect game – an empirical study of game modifications","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Digital Games and Media","field":"Social Sciences","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Queen's University","funders":"","keywords":"Nexus (standard); Game Developer; Video game development; Game design; Empirical research; Game mechanics; Expectancy theory; Video game; Game design document","score_opus":0.06416992483423542,"score_gpt":0.3518687718725928,"score_spread":0.28769884703835735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3013163251","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9762695,0.00013351822,0.013232665,0.00017200431,0.000010948174,0.0001722852,0.000067161316,0.000050769955,0.009891168],"genre_scores_gemma":[0.9948885,0.000038310736,0.0042138374,0.000022998973,0.000002046844,0.00004308098,0.000034305842,0.000021207114,0.0007358187],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9956181,0.0026569213,0.00024369611,0.000470102,0.0007964028,0.00021487757],"domain_scores_gemma":[0.8993673,0.0766148,0.0074302056,0.011930711,0.0031782705,0.0014788492],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059573967,0.0005861019,0.0005028982,0.0012980545,0.0014943058,0.0028233822,0.0017419163,0.000912057,0.004130852],"category_scores_gemma":[0.09964481,0.0006582155,0.00042044473,0.001116178,0.0049497634,0.0068147704,0.0021141928,0.0020666616,0.0002811599],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022440408,0.005467924,0.27772966,0.001083903,0.00048103387,0.00073173153,0.062006734,0.035031818,0.014707304,0.38277927,0.0036309997,0.2141056],"study_design_scores_gemma":[0.00045368992,0.0059138224,0.2984535,0.000551984,0.00055471965,0.0016769848,0.060766734,0.25375068,0.018871007,0.32035854,0.038303208,0.00034504608],"about_ca_topic_score_codex":0.0040477547,"about_ca_topic_score_gemma":0.004860885,"teacher_disagreement_score":0.0059573967,"about_ca_system_score_codex":0.0013862718,"about_ca_system_score_gemma":0.0013384867,"threshold_uncertainty_score":0.03150612},"labels":[],"label_agreement":null},{"id":"W3016381677","doi":"10.1007/s10664-020-09818-7","title":"How software engineering research aligns with design science: a review","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":69,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Lunds Universitet","keywords":"Design science; Novelty; Software; Software design; Set (abstract data type); Design science research; Social software engineering","score_opus":0.13199070938243,"score_gpt":0.34603854400817013,"score_spread":0.21404783462574015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3016381677","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005973971,0.99475753,0.0011130262,0.001943805,0.00029336414,0.000015166812,0.000023898821,0.0000123003,0.0012435855],"genre_scores_gemma":[0.0070306,0.9904152,0.0015778854,0.0005744628,0.00019017809,0.000027057878,0.00003664567,0.000011390255,0.00013654685],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9893679,0.0039433995,0.0024500028,0.0010305019,0.0028736282,0.0003345046],"domain_scores_gemma":[0.88472307,0.09495442,0.006320593,0.00146281,0.011673176,0.0008659212],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.012283135,0.00078340847,0.0014929133,0.026052626,0.00108776,0.0062781554,0.0013686899,0.0023177466,0.0020237505],"category_scores_gemma":[0.055003475,0.0007978031,0.0013802015,0.030858424,0.002301346,0.006555707,0.0017585318,0.0021611801,0.00072365766],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000615774,0.000059389324,0.0030266896,0.12180027,0.0005668415,0.00032850975,0.0029495512,0.0013578305,0.0014906636,0.027672239,0.018893244,0.8217932],"study_design_scores_gemma":[0.00001989145,0.00012780761,0.007244172,0.2412904,0.0011532959,0.0012684966,0.0032680752,0.00062591885,0.0010288319,0.021071356,0.72280425,0.000097569944],"about_ca_topic_score_codex":0.004069689,"about_ca_topic_score_gemma":0.007849332,"teacher_disagreement_score":0.98771685,"about_ca_system_score_codex":0.004140418,"about_ca_system_score_gemma":0.010635252,"threshold_uncertainty_score":0.06496018},"labels":[],"label_agreement":null},{"id":"W3016717485","doi":"10.1007/s10664-020-09814-x","title":"Using machine learning to assist with the selection of security controls during security assessment","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Information and Cyber Security","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Operationalization; Computer science; Security information and event management; Security controls; Computer security model; Computer security; Context (archaeology); Security service; Security testing; Cloud computing security; Security engineering; Security through obscurity; Security domain; Standard of Good Practice; Security convergence; Information security; Risk analysis (engineering); Software security assurance; Control (management); Artificial intelligence; Business; Cloud computing; Network security policy","score_opus":0.01568697951433107,"score_gpt":0.26038288969559326,"score_spread":0.2446959101812622,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3016717485","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35455605,0.00027867584,0.62729204,0.0008791965,0.00011696455,0.0003079044,0.00032085672,0.0049342453,0.011314031],"genre_scores_gemma":[0.9005856,0.00004510627,0.09795949,0.000069102716,0.000013217469,0.000051393934,0.00014345137,0.000057931014,0.0010747649],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981345,0.0009907791,0.00013604847,0.0002626698,0.00034763536,0.0001283413],"domain_scores_gemma":[0.98828095,0.0076938216,0.0011942589,0.0006420561,0.0019165769,0.0002722988],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00346741,0.0007287365,0.0005030538,0.0019740616,0.0005725651,0.0016639513,0.00069219334,0.0008384075,0.0033921266],"category_scores_gemma":[0.02062868,0.00025391104,0.000261142,0.00058900495,0.00033090965,0.0014334121,0.0007324985,0.0010642892,0.0010180983],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008597159,0.0011391831,0.071217656,0.00019910277,0.00014336049,0.00019510007,0.0008409146,0.113423005,0.025242077,0.004393025,0.006040513,0.77630633],"study_design_scores_gemma":[0.000045815617,0.00017154294,0.013538145,0.000056406327,0.00003410714,0.000068839196,0.00022230383,0.96002036,0.017604187,0.0064378376,0.001759151,0.000041355925],"about_ca_topic_score_codex":0.0038334175,"about_ca_topic_score_gemma":0.006346856,"teacher_disagreement_score":0.0038334175,"about_ca_system_score_codex":0.0006019461,"about_ca_system_score_gemma":0.0012711649,"threshold_uncertainty_score":0.018337667},"labels":[],"label_agreement":null},{"id":"W3017078585","doi":"10.1007/s10664-020-09806-x","title":"Preface to the special issue on program comprehension","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Program comprehension; Computer science; Comprehension; Programming language; Software","score_opus":0.03621576006486358,"score_gpt":0.27980399103285897,"score_spread":0.2435882309679954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3017078585","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00050000043,0.008299265,0.0026661986,0.03892591,0.9286954,0.00028377256,0.0011296971,0.00072058523,0.018779255],"genre_scores_gemma":[0.00278488,0.009865848,0.0019305749,0.014555091,0.8339548,0.00044471704,0.0031015833,0.0012421901,0.13212025],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9968303,0.00047175444,0.0003579283,0.0006601821,0.0013941997,0.00028557194],"domain_scores_gemma":[0.9754479,0.004405863,0.0012044695,0.0012242203,0.012194289,0.0055232663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040820017,0.0025258185,0.0025300388,0.0054356134,0.0024654593,0.009701378,0.0021409385,0.0037556188,0.18624352],"category_scores_gemma":[0.01849412,0.0007597289,0.0020459264,0.0028573968,0.0009301862,0.005298232,0.00391132,0.006320052,0.1085439],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026697302,0.000027570468,0.00006410185,0.00017424623,0.0000063334774,0.00003731316,0.000036641883,0.00003249698,0.00019610452,0.0003722095,0.9789508,0.02007557],"study_design_scores_gemma":[0.000018997347,0.000072589224,0.0006736416,0.00031562426,0.000012308273,0.0001473427,0.00007652894,0.00011280933,0.00012864263,0.0014768556,0.99694616,0.000018503446],"about_ca_topic_score_codex":0.00077136635,"about_ca_topic_score_gemma":0.0012559593,"teacher_disagreement_score":0.18624352,"about_ca_system_score_codex":0.0022824327,"about_ca_system_score_gemma":0.0026491676,"threshold_uncertainty_score":0.62304664},"labels":[],"label_agreement":null},{"id":"W3018447383","doi":"10.1007/s10664-020-09819-6","title":"What do Programmers Discuss about Deep Learning Frameworks","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":89,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Deep learning; Workflow; Computer science; Latent Dirichlet allocation; Leverage (statistics); Artificial intelligence; Topic model; Data science; Profiling (computer programming); World Wide Web; Machine learning","score_opus":0.01947834175930554,"score_gpt":0.2802885285859658,"score_spread":0.2608101868266603,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3018447383","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03961806,0.025653223,0.16569036,0.6940899,0.005777477,0.000049728835,0.00029004982,0.000631028,0.06820028],"genre_scores_gemma":[0.69162405,0.03574362,0.097078905,0.13169253,0.01473747,0.00019993496,0.00060862734,0.0019149282,0.02639988],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.9898347,0.004696473,0.00047088417,0.0011224025,0.0027704025,0.0011050804],"domain_scores_gemma":[0.9206953,0.056469396,0.004114738,0.0050578797,0.010067845,0.0035948043],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01687423,0.00065518566,0.0006053128,0.0022965523,0.0033350934,0.010046354,0.0021529181,0.005931432,0.007407367],"category_scores_gemma":[0.11264685,0.00073939044,0.00073387666,0.0028980966,0.0067638187,0.031993896,0.0025876875,0.008907746,0.0015571446],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001883414,0.00013982429,0.011869071,0.0011793923,0.00013071358,0.00027281343,0.0061049946,0.0028189027,0.0013386308,0.5466093,0.090537645,0.33881035],"study_design_scores_gemma":[0.000039018578,0.00006488232,0.0035754961,0.002943536,0.000108349705,0.00079435884,0.01025494,0.0052162986,0.0028133434,0.6622452,0.31184426,0.00010026101],"about_ca_topic_score_codex":0.0028736605,"about_ca_topic_score_gemma":0.003119077,"teacher_disagreement_score":0.01687423,"about_ca_system_score_codex":0.0022939001,"about_ca_system_score_gemma":0.003750404,"threshold_uncertainty_score":0.08924049},"labels":[],"label_agreement":null},{"id":"W3021573492","doi":"10.1007/s10664-021-09956-6","title":"On systematically building a controlled natural language for functional requirements","year":2021,"lang":"en","type":"preprint","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg","keywords":"Computer science; Domain (mathematical analysis); Natural language; Vagueness; Context (archaeology); Ambiguity; Flexibility (engineering); Quality (philosophy); Popularity; Grammar; Natural language processing; Artificial intelligence; Programming language; Linguistics; Psychology","score_opus":0.034998506852739225,"score_gpt":0.31997698353604714,"score_spread":0.2849784766833079,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3021573492","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009754392,0.00016735354,0.9805519,0.0012521847,0.000036191082,0.0023532836,0.00071886176,0.0006756776,0.00449007],"genre_scores_gemma":[0.03733023,0.00014787748,0.9580964,0.00029986753,0.000014491738,0.0021791845,0.0011997448,0.00017063793,0.00056165253],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9471646,0.037743956,0.0037020212,0.0033544223,0.0074230437,0.0006119171],"domain_scores_gemma":[0.7570104,0.20154248,0.009327099,0.013377129,0.01779956,0.000943272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04404705,0.0016032204,0.00073986617,0.005168967,0.002259487,0.0046704304,0.0031339612,0.0016218092,0.004979904],"category_scores_gemma":[0.13591447,0.0013615399,0.0020964316,0.002665836,0.0067081195,0.009364893,0.0059496984,0.0035096135,0.0014420211],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019913197,0.0007623284,0.0069116415,0.0051422333,0.00012027992,0.00092636107,0.038829446,0.039228097,0.027528943,0.5508078,0.009264402,0.32027936],"study_design_scores_gemma":[0.0004144345,0.0010709916,0.0042398265,0.006257536,0.00017561455,0.0016940432,0.022972494,0.23294689,0.03899611,0.41210905,0.27860978,0.0005132936],"about_ca_topic_score_codex":0.008099574,"about_ca_topic_score_gemma":0.01650586,"teacher_disagreement_score":0.04404705,"about_ca_system_score_codex":0.00431125,"about_ca_system_score_gemma":0.012778916,"threshold_uncertainty_score":0.2329458},"labels":[],"label_agreement":null},{"id":"W3024356476","doi":"10.1007/s10664-020-09878-9","title":"On the time-based conclusion stability of cross-project defect prediction models","year":2020,"lang":"en","type":"preprint","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Stability (learning theory); Computer science; Software; Predictive modelling; Product (mathematics); Limit (mathematics); Data mining; Time limit; Analytics; Data science; Empirical research; Econometrics; Reliability engineering; Statistics; Machine learning; Mathematics; Engineering; Systems engineering","score_opus":0.07151071104906312,"score_gpt":0.31989840498568256,"score_spread":0.24838769393661944,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3024356476","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37764737,0.0025657122,0.6092844,0.0035184233,0.0002443677,0.000094051364,0.0005944589,0.0006364219,0.0054148664],"genre_scores_gemma":[0.9808966,0.00041313548,0.015994022,0.00022463051,0.0001351492,0.00004480467,0.0005545118,0.00016740631,0.0015696423],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9942,0.003392561,0.00023518606,0.0011983004,0.0006909049,0.0002830732],"domain_scores_gemma":[0.7162458,0.2610842,0.0064616073,0.006342258,0.008508021,0.0013581788],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.027729636,0.0009094398,0.0016086847,0.0022694631,0.000936244,0.0028155996,0.0024940763,0.002226517,0.003733435],"category_scores_gemma":[0.16730647,0.0006561574,0.0012672795,0.0013499568,0.0020213008,0.004309156,0.0025868302,0.0039507058,0.00047189195],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012034791,0.0002073446,0.02870645,0.0002430945,0.00053214683,0.00026716513,0.0004770511,0.83459824,0.002213557,0.06482403,0.003430249,0.06329714],"study_design_scores_gemma":[0.000009452336,0.00003723403,0.0013962862,0.0000232581,0.000030152083,0.000017777978,0.000028633829,0.98668706,0.00032832622,0.011322481,0.00010971967,0.00000970578],"about_ca_topic_score_codex":0.0069598425,"about_ca_topic_score_gemma":0.0035221933,"teacher_disagreement_score":0.97227037,"about_ca_system_score_codex":0.0017083236,"about_ca_system_score_gemma":0.001323066,"threshold_uncertainty_score":0.14665008},"labels":[],"label_agreement":null},{"id":"W3028200206","doi":"10.1007/s10664-020-09837-4","title":"Do code review measures explain the incidence of post-release defects?","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Operationalization; Code (set theory); Computer science; Replication (statistics); Code review; Replicate; Variable (mathematics); Empirical research; Contrast (vision); Econometrics; Statistics; Software; Software quality; Artificial intelligence; Mathematics; Software development; Programming language","score_opus":0.042557336039330304,"score_gpt":0.29470602258305645,"score_spread":0.25214868654372613,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3028200206","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99370366,0.0012042754,0.0015412857,0.0007922967,0.000048163292,0.00002815654,0.0008260564,0.000067120476,0.0017888768],"genre_scores_gemma":[0.99890053,0.000112194626,0.00022055203,0.000068714326,0.000023710196,0.000009732918,0.00029750384,0.00002789455,0.00033910535],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99234444,0.0022400753,0.0011326843,0.001089467,0.002406876,0.00078647124],"domain_scores_gemma":[0.5384732,0.24040405,0.18205297,0.017546702,0.01634718,0.005175964],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.011275735,0.0005604706,0.0006285384,0.0049251085,0.00042395227,0.0021399518,0.0016613622,0.0018356931,0.004334835],"category_scores_gemma":[0.18100414,0.0005366016,0.0010331775,0.003806828,0.0012024358,0.0030761212,0.00095748954,0.0017888322,0.0010394948],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010748851,0.00008627708,0.99232,0.000054815482,0.0001967927,0.00004091048,0.00017511536,0.00026760763,0.00020183757,0.0001790557,0.00035618414,0.0060139503],"study_design_scores_gemma":[0.000008596455,0.00010447547,0.9978309,0.00003765348,0.00006558553,0.000080918,0.00018170472,0.00086086255,0.00023345857,0.00026948986,0.000315886,0.000010359834],"about_ca_topic_score_codex":0.005785036,"about_ca_topic_score_gemma":0.009438024,"teacher_disagreement_score":0.9887243,"about_ca_system_score_codex":0.0008661799,"about_ca_system_score_gemma":0.0013501208,"threshold_uncertainty_score":0.05963254},"labels":[],"label_agreement":null},{"id":"W3030475425","doi":"10.1007/s10664-020-09858-z","title":"The who, what, how of software engineering research: a socio-technical framework","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":90,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Social software engineering; Software peer review; Software development; Empirical research; Beneficiary; Personal software process; Software","score_opus":0.08362109294565391,"score_gpt":0.3358897120986377,"score_spread":0.2522686191529838,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3030475425","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014901354,0.04415028,0.27939898,0.5593503,0.002646405,0.00067794044,0.0003853255,0.00024101937,0.09824842],"genre_scores_gemma":[0.7266641,0.026902106,0.19734596,0.035757244,0.00487037,0.0017037872,0.00022566876,0.00032649166,0.006204146],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9352764,0.04800886,0.0038568522,0.0032150398,0.007889889,0.0017529582],"domain_scores_gemma":[0.82638884,0.14634427,0.0058663087,0.005801778,0.008385471,0.007213371],"candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.07468481,0.0019622291,0.0032454992,0.016740913,0.011397168,0.027050836,0.0043616593,0.01591273,0.004707928],"category_scores_gemma":[0.06251721,0.002234474,0.0014353769,0.0104379365,0.15390402,0.05962122,0.0108777005,0.013735725,0.0008919235],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000008446341,0.000057089877,0.00063568127,0.00020969882,0.000015067373,0.00008293033,0.0041568684,0.00027011326,0.00010984682,0.9866121,0.0017993079,0.0060429582],"study_design_scores_gemma":[0.00002131922,0.00002477069,0.00045928796,0.0006899047,0.000012846834,0.00013792298,0.006234747,0.0011156532,0.00013428609,0.96859556,0.022544445,0.000029252798],"about_ca_topic_score_codex":0.0057600774,"about_ca_topic_score_gemma":0.005149952,"teacher_disagreement_score":0.9886028,"about_ca_system_score_codex":0.013190762,"about_ca_system_score_gemma":0.02718721,"threshold_uncertainty_score":0.39497578},"labels":[],"label_agreement":null},{"id":"W3036270494","doi":"10.1007/s10664-021-09951-x","title":"Lags in the release, adoption, and propagation of npm vulnerability fixes","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Information and Cyber Security","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Japan Society for the Promotion of Science","keywords":"Vulnerability (computing); Empirical research; Vulnerability assessment; Software release life cycle; Software","score_opus":0.014491196281586089,"score_gpt":0.246677101234731,"score_spread":0.2321859049531449,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3036270494","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9964606,0.00021359592,0.0014496873,0.00013612611,0.0000127631,0.000022356551,0.00030273994,0.00009207933,0.0013101548],"genre_scores_gemma":[0.99828243,0.000093815535,0.00087037147,0.000022825237,0.000009412539,0.00001531314,0.000309985,0.000025484433,0.00037037907],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99504745,0.0010055862,0.00058247615,0.0011024337,0.0018050458,0.00045717074],"domain_scores_gemma":[0.8436761,0.08021709,0.050201606,0.010103846,0.012427659,0.003373737],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061271903,0.0003371657,0.00023859016,0.0027990497,0.0005388293,0.0017931849,0.00067855185,0.0005433052,0.0025029464],"category_scores_gemma":[0.080709,0.00043279462,0.0002745829,0.0021456103,0.0008582196,0.0022529252,0.0012504488,0.0018507976,0.0005273364],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014748893,0.00011844731,0.9603855,0.00012435742,0.00006503028,0.00045126837,0.0034599386,0.0020287759,0.0032216082,0.0009985489,0.00088930165,0.028109673],"study_design_scores_gemma":[0.000007086343,0.00015727161,0.99010414,0.000061082006,0.000022012498,0.00031928346,0.0021020256,0.0035567817,0.0012026455,0.0005091762,0.001924584,0.00003393351],"about_ca_topic_score_codex":0.0048001097,"about_ca_topic_score_gemma":0.006686732,"teacher_disagreement_score":0.0061271903,"about_ca_system_score_codex":0.00088035746,"about_ca_system_score_gemma":0.0007297366,"threshold_uncertainty_score":0.032404065},"labels":[],"label_agreement":null},{"id":"W3048902333","doi":"10.1007/s10664-020-09822-x","title":"A study of the performance of general compressors on log files","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University","funders":"","keywords":"Compression ratio; Gas compressor; Lossless compression; Data compression; Computer science; Compression (physics); Volume (thermodynamics); Database; Real-time computing; Algorithm; Engineering; Physics","score_opus":0.026760541705675916,"score_gpt":0.2510870800200216,"score_spread":0.2243265383143457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3048902333","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99913955,0.00005132212,0.00034419654,0.000023467597,0.000002067321,0.0000095139585,0.000053822525,0.000030590796,0.00034541372],"genre_scores_gemma":[0.9990158,0.00007118193,0.00038093256,0.0000052678497,0.0000048980814,0.000003981189,0.00010151689,0.000008373658,0.0004079857],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99946636,0.00021595861,0.000025097968,0.00005357078,0.00016691475,0.00007207939],"domain_scores_gemma":[0.96951723,0.02697561,0.0007400287,0.0009811026,0.0014685416,0.00031751394],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010457216,0.00045058245,0.00036806523,0.000964703,0.0005808926,0.0004977358,0.0006194394,0.00046261217,0.0024927587],"category_scores_gemma":[0.015943741,0.00022371992,0.0002488756,0.001593869,0.00057472946,0.0010021027,0.00022629862,0.0005664285,0.00019065876],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.016206622,0.006025758,0.35696915,0.001101433,0.00056412775,0.0015166701,0.0039883684,0.21187945,0.14529149,0.0063678967,0.0063411715,0.24374792],"study_design_scores_gemma":[0.00037337365,0.010178993,0.43012917,0.000045422195,0.00022183197,0.0005728747,0.002470238,0.47700766,0.07515846,0.0017168826,0.0020401962,0.00008490703],"about_ca_topic_score_codex":0.013089511,"about_ca_topic_score_gemma":0.007976347,"teacher_disagreement_score":0.013089511,"about_ca_system_score_codex":0.0007394376,"about_ca_system_score_gemma":0.0005022701,"threshold_uncertainty_score":0.026026666},"labels":[],"label_agreement":null},{"id":"W3080471220","doi":"10.1007/s10664-020-09840-9","title":"An empirical study of the characteristics of popular Minecraft mods","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Queen's University","funders":"","keywords":"Popularity; Remuneration; Context (archaeology); Mod; Empirical research; Empirical evidence","score_opus":0.0470913351693498,"score_gpt":0.312429986715034,"score_spread":0.2653386515456842,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3080471220","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99721503,0.000021564083,0.00019830056,0.000038695005,0.0000015947137,0.000009952277,0.00012125616,0.0000053585436,0.0023882245],"genre_scores_gemma":[0.99833566,0.00002034982,0.00027065098,0.000012300945,0.0000018930881,0.000007445973,0.00020977811,0.000006044026,0.001135821],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99942636,0.00014280448,0.000051326377,0.000085753265,0.00021566349,0.0000780957],"domain_scores_gemma":[0.9814175,0.007215007,0.007218073,0.00084752456,0.0019492132,0.0013526436],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008346357,0.0001391634,0.00014727483,0.0025073981,0.00071978674,0.0011565958,0.00046052603,0.00038443645,0.0049176477],"category_scores_gemma":[0.015149087,0.00017783279,0.00010146376,0.0023023272,0.00071430847,0.0013126231,0.00066853315,0.000515586,0.0008807451],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014939299,0.00021359981,0.9761461,0.000031969506,0.000020960539,0.00009097813,0.0032894912,0.0001621653,0.0008884499,0.0021817256,0.0008249088,0.016000168],"study_design_scores_gemma":[0.000008415122,0.00009534905,0.9869876,0.00001655994,0.000008067276,0.00024720788,0.006821299,0.0018309235,0.00033874204,0.0005796934,0.0030557823,0.000010302382],"about_ca_topic_score_codex":0.0035001321,"about_ca_topic_score_gemma":0.0082433615,"teacher_disagreement_score":0.0049176477,"about_ca_system_score_codex":0.0005196376,"about_ca_system_score_gemma":0.0004083418,"threshold_uncertainty_score":0.01645118},"labels":[],"label_agreement":null},{"id":"W3081943439","doi":"10.1007/s10664-020-09863-2","title":"CROKAGE: effective solution recommendation for programming tasks by leveraging crowd knowledge","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Computer science; Leverage (statistics); Code (set theory); Information retrieval; Task (project management); Relevance (law); Programming language; Artificial intelligence","score_opus":0.03546418867217926,"score_gpt":0.3080167954479544,"score_spread":0.27255260677577514,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3081943439","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16161682,0.006738643,0.75025326,0.0019930042,0.0012420157,0.001461876,0.0048623495,0.052109867,0.019722156],"genre_scores_gemma":[0.4182077,0.00083497394,0.5584208,0.00078086334,0.00028258134,0.0006767238,0.0077180834,0.0010032548,0.012075054],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99830544,0.00039678058,0.00006344386,0.00054626475,0.0005312573,0.00015676426],"domain_scores_gemma":[0.99722666,0.0013028004,0.00014748119,0.0006167785,0.00048095433,0.0002252318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017689326,0.0020507914,0.0016616777,0.0056689293,0.001287634,0.0015628159,0.0023172451,0.0028520324,0.00532426],"category_scores_gemma":[0.009762625,0.0007053831,0.001127334,0.002604997,0.0006490595,0.0034459927,0.0025669083,0.0018348554,0.0026239667],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016225956,0.002131168,0.010572235,0.0007313925,0.000523725,0.00027711567,0.0003885941,0.06307398,0.014436991,0.0050375033,0.091353066,0.8098516],"study_design_scores_gemma":[0.0002480451,0.0002829508,0.0018632733,0.00006461775,0.000113322705,0.00011702862,0.00014487894,0.9708827,0.005056873,0.009590792,0.011568202,0.00006726633],"about_ca_topic_score_codex":0.016550018,"about_ca_topic_score_gemma":0.039880674,"teacher_disagreement_score":0.016550018,"about_ca_system_score_codex":0.0008308554,"about_ca_system_score_gemma":0.0020176799,"threshold_uncertainty_score":0.032907367},"labels":[],"label_agreement":null},{"id":"W3082966820","doi":"10.1007/s10664-021-10005-5","title":"Understanding peer review of software engineering papers","year":2021,"lang":"en","type":"preprint","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Novelty; Quality (philosophy); Computer science; Technical peer review; Peer review; Software technical review; Psychology; Medical education; Software; Engineering ethics; Software quality; Software development; Engineering; Medicine; Political science; Social psychology","score_opus":0.10291582984899653,"score_gpt":0.3172505489139484,"score_spread":0.21433471906495186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3082966820","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16772424,0.06764422,0.20906317,0.18802309,0.014767412,0.0011476023,0.0023949435,0.0028438638,0.34639147],"genre_scores_gemma":[0.90028065,0.015762819,0.03147718,0.005168047,0.008312127,0.00035692626,0.0021431227,0.0011278685,0.035371337],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.898334,0.057304997,0.005584836,0.0056114956,0.030560583,0.0026040229],"domain_scores_gemma":[0.33479303,0.47737786,0.039020002,0.030546512,0.110902734,0.0073598633],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06389796,0.0007631439,0.001471231,0.010776294,0.003957869,0.02156978,0.0027113883,0.0058126887,0.023699166],"category_scores_gemma":[0.5207444,0.00084052805,0.0009225585,0.008372445,0.0038793301,0.025347808,0.0054076,0.0035289226,0.0054954733],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042761868,0.00026156858,0.029271258,0.0028173807,0.0005214743,0.0010794471,0.022766251,0.0047422787,0.0020155269,0.38106373,0.20886226,0.3461712],"study_design_scores_gemma":[0.00019267293,0.00018990683,0.021116583,0.0017135747,0.00032327414,0.000727453,0.010026106,0.015496283,0.0024321128,0.5948268,0.35273582,0.00021953175],"about_ca_topic_score_codex":0.003804906,"about_ca_topic_score_gemma":0.0031999422,"teacher_disagreement_score":0.93610203,"about_ca_system_score_codex":0.004606027,"about_ca_system_score_gemma":0.010072398,"threshold_uncertainty_score":0.3379287},"labels":[],"label_agreement":null},{"id":"W3084301791","doi":"10.1007/s10664-020-09864-1","title":"Automated demarcation of requirements in textual specifications: a machine learning-based approach","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"H2020 European Research Council; Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg; European Commission","keywords":"Computer science; Software requirements specification; Requirements analysis; Formal specification; Markup language; System requirements specification; Simple (philosophy); Identifier; Task (project management); Requirements engineering; Variety (cybernetics); Software engineering; Enforcement; Artificial intelligence; Programming language; XML; Systems engineering; Engineering; Software; World Wide Web","score_opus":0.07854626844973958,"score_gpt":0.29834454275323014,"score_spread":0.21979827430349055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3084301791","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10854295,0.0023529162,0.8090057,0.0017667296,0.00018626933,0.0009140429,0.008728311,0.062326413,0.006176751],"genre_scores_gemma":[0.19577621,0.00033608056,0.7736245,0.0004891765,0.00005025349,0.00031345323,0.02630532,0.0006282049,0.0024768836],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99300086,0.0022299304,0.0008781007,0.0019161309,0.0016988214,0.000276157],"domain_scores_gemma":[0.9773111,0.0115756225,0.0028244376,0.002538152,0.0052999235,0.0004507943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039475174,0.0020233488,0.00097248005,0.008181075,0.00085891125,0.0022700203,0.0030005435,0.0020728745,0.003406782],"category_scores_gemma":[0.018711718,0.00053241145,0.0017006601,0.0034147806,0.00079465396,0.0027976593,0.001743358,0.0025242346,0.0034167448],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039564836,0.00065780664,0.013836767,0.001348865,0.00015588742,0.00064288545,0.00078295654,0.051770534,0.022070797,0.003925346,0.03497245,0.86944014],"study_design_scores_gemma":[0.00008026563,0.00016564659,0.005569999,0.00026458263,0.000076307995,0.000500647,0.0008384772,0.92949206,0.027664803,0.0086333165,0.026644,0.00006994153],"about_ca_topic_score_codex":0.011739465,"about_ca_topic_score_gemma":0.020573176,"teacher_disagreement_score":0.011739465,"about_ca_system_score_codex":0.0021786937,"about_ca_system_score_gemma":0.0026722853,"threshold_uncertainty_score":0.023342252},"labels":[],"label_agreement":null},{"id":"W3084421431","doi":"10.1007/s10664-020-09852-5","title":"Code cloning in smart contracts: a case study on verified contracts from the Ethereum blockchain platform","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Queen's University","funders":"Japan Society for the Promotion of Science; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Smart contract; clone (Java method); Blockchain; Cryptocurrency; Code (set theory); Source code; Backward compatibility; Cloning (programming); Computer science; Order (exchange); Computer security; Business; Operating system; Programming language; Finance","score_opus":0.05194985740488874,"score_gpt":0.2930333489933576,"score_spread":0.24108349158846884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3084421431","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9886841,0.00010938287,0.0079607945,0.00030257014,0.000007633348,0.00008449624,0.000121736324,0.0001009426,0.0026282615],"genre_scores_gemma":[0.99048924,0.00008949842,0.0072331987,0.00003925843,0.000004545863,0.0000280409,0.00021523675,0.000053061263,0.0018479808],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99402195,0.0025237082,0.0002840717,0.00043724658,0.0021342488,0.0005987834],"domain_scores_gemma":[0.9183595,0.06472749,0.004260191,0.0071622888,0.004389409,0.0011011923],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007961136,0.00035623394,0.00036305474,0.0011772616,0.0022429577,0.0015832109,0.0013089684,0.002449642,0.0026936734],"category_scores_gemma":[0.043709632,0.00039610406,0.00035859045,0.0021202576,0.0027242168,0.003820108,0.0017657435,0.0017386401,0.00042997205],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026484157,0.0045856107,0.4216872,0.0010524819,0.00026182635,0.03457987,0.030398933,0.13247208,0.026839,0.09716429,0.008586897,0.23972337],"study_design_scores_gemma":[0.00043345668,0.0021107483,0.16759363,0.00069342455,0.00026062308,0.012918103,0.02598399,0.6273397,0.062122636,0.053742886,0.046515025,0.00028581457],"about_ca_topic_score_codex":0.008729538,"about_ca_topic_score_gemma":0.010444273,"teacher_disagreement_score":0.008729538,"about_ca_system_score_codex":0.0016149853,"about_ca_system_score_gemma":0.0029221964,"threshold_uncertainty_score":0.042103052},"labels":[],"label_agreement":null},{"id":"W3091970108","doi":"10.1007/s10664-020-09851-6","title":"Publish or perish, but do not forget your software artifacts","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Deutsche Forschungsgemeinschaft; Deutscher Akademischer Austauschdienst","keywords":"Artifact (error); Publish or perish; Computer science; Publication; Context (archaeology); Replication (statistics); Data science; Software; Empirical research; Open science; Software engineering; World Wide Web; Publishing; Artificial intelligence; Political science","score_opus":0.06981769528065791,"score_gpt":0.2983438529525711,"score_spread":0.2285261576719132,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3091970108","genre_codex":"other","genre_gemma":"commentary","domain_codex":null,"domain_gemma":"incentives","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":"incentives","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25101632,0.055414543,0.09041624,0.19397044,0.054083925,0.0015920047,0.051693123,0.011358517,0.29045492],"genre_scores_gemma":[0.7279933,0.029746272,0.073614255,0.021426616,0.0232576,0.0010488536,0.03157077,0.0056645363,0.085677885],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.95153445,0.01671341,0.005835456,0.0033123859,0.021276671,0.0013275832],"domain_scores_gemma":[0.5170412,0.22455345,0.06974537,0.104406245,0.07098224,0.013271527],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.039264724,0.0005856666,0.00087825913,0.01326257,0.0031891295,0.014657185,0.0018958541,0.0020797877,0.038384575],"category_scores_gemma":[0.33363926,0.0005147593,0.0011528082,0.02374813,0.0029277757,0.013517389,0.0059717502,0.0025810427,0.025200913],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042851432,0.00023368324,0.07632587,0.0066062603,0.0004945968,0.0011080619,0.0072791986,0.00063848204,0.0026832996,0.044304773,0.41170305,0.4481942],"study_design_scores_gemma":[0.00008438859,0.0001607789,0.028728208,0.0028115918,0.00015753704,0.0008668721,0.0028750626,0.0005024544,0.0021267787,0.036100093,0.92548233,0.000103776474],"about_ca_topic_score_codex":0.0012807144,"about_ca_topic_score_gemma":0.0021235896,"teacher_disagreement_score":0.96073526,"about_ca_system_score_codex":0.0018568309,"about_ca_system_score_gemma":0.0052235415,"threshold_uncertainty_score":0.20765418},"labels":[],"label_agreement":null},{"id":"W3092232764","doi":"10.1007/s10664-020-09916-6","title":"What makes a popular academic AI repository?","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"License; Publishing; Set (abstract data type); Software; Empirical research; Inclusion (mineral)","score_opus":0.10157218383589905,"score_gpt":0.39141405928048384,"score_spread":0.2898418754445848,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3092232764","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13703336,0.012440872,0.011201485,0.51660424,0.006220339,0.00015143695,0.0016721844,0.00228336,0.3123927],"genre_scores_gemma":[0.8473715,0.0071977703,0.013984797,0.033795476,0.010432039,0.00024905964,0.0023843884,0.0026585164,0.08192633],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9549057,0.018286234,0.0024822364,0.003982393,0.01554061,0.004802835],"domain_scores_gemma":[0.780905,0.057379767,0.021400437,0.02337625,0.062123425,0.054815166],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.035920486,0.0006907017,0.0012407971,0.019610377,0.013580119,0.062065125,0.006239618,0.00855393,0.03766964],"category_scores_gemma":[0.14865848,0.0008511924,0.0007082694,0.024021162,0.01147917,0.06719071,0.013307243,0.005163624,0.016746651],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005596177,0.0005684843,0.03839129,0.00075081305,0.00018175328,0.0010374084,0.014623289,0.00064054015,0.0011163593,0.31440464,0.37909752,0.24862817],"study_design_scores_gemma":[0.00013032636,0.00020726252,0.015196564,0.0013432052,0.000111958165,0.0015431085,0.046880826,0.002192554,0.0013978617,0.15900141,0.7717151,0.00027975944],"about_ca_topic_score_codex":0.006027469,"about_ca_topic_score_gemma":0.009431615,"teacher_disagreement_score":0.9803896,"about_ca_system_score_codex":0.008062792,"about_ca_system_score_gemma":0.013717435,"threshold_uncertainty_score":0.18996793},"labels":[],"label_agreement":null},{"id":"W3096162789","doi":"10.1007/s10664-020-09874-z","title":"A feature location approach for mapping application features extracted from crowd-based screencasts to source code","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Concordia University","funders":"","keywords":"Computer science; Source code; Codebase; Set (abstract data type); Program comprehension; Software; Workflow; Code (set theory); Information retrieval; Multimedia; World Wide Web; Human–computer interaction; Database; Software system; Programming language","score_opus":0.03464601524693783,"score_gpt":0.2754876655325143,"score_spread":0.24084165028557647,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3096162789","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11183899,0.00033698356,0.8720528,0.00019313321,0.00013047237,0.00022448413,0.0023087158,0.0077651152,0.005149411],"genre_scores_gemma":[0.6687675,0.00018033282,0.32152003,0.000062972445,0.00008685599,0.00028520063,0.0027853923,0.0002760505,0.0060356217],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9994062,0.000083624225,0.000024173358,0.0002121068,0.00019099619,0.00008285067],"domain_scores_gemma":[0.9987962,0.0003151611,0.0001440806,0.00022562545,0.00044435685,0.000074491705],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000439507,0.0007582382,0.0006053616,0.0042827753,0.0005083641,0.00092437014,0.0007670635,0.0009168015,0.0024411064],"category_scores_gemma":[0.0030092201,0.00025070252,0.0006225725,0.0026375942,0.0003323415,0.0010931528,0.0015770827,0.00059852784,0.0020721692],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007864829,0.0005829409,0.02861005,0.00030823867,0.00020723534,0.0008516751,0.0016142591,0.02314644,0.097427525,0.0055356016,0.015761767,0.8251678],"study_design_scores_gemma":[0.00009001906,0.00047225328,0.06280711,0.00007683021,0.00021642569,0.000962052,0.0019682073,0.85438555,0.04650876,0.012010929,0.020331426,0.00017047478],"about_ca_topic_score_codex":0.010033718,"about_ca_topic_score_gemma":0.014169695,"teacher_disagreement_score":0.010033718,"about_ca_system_score_codex":0.00038660757,"about_ca_system_score_gemma":0.0007569639,"threshold_uncertainty_score":0.019950628},"labels":[],"label_agreement":null},{"id":"W3118322826","doi":"10.1007/s10664-020-09921-9","title":"ID-correspondence: a measure for detecting evolutionary coupling","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Identifier; False positive paradox; Computer science; Similarity (geometry); Measure (data warehouse); Artificial intelligence; Coupling (piping); Data mining; Natural language processing; Machine learning; Programming language; Engineering","score_opus":0.032689246102338025,"score_gpt":0.29433416246353283,"score_spread":0.2616449163611948,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3118322826","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25922307,0.00084074924,0.7190608,0.00024741283,0.00021937632,0.0002827004,0.0023541895,0.0036121244,0.014159644],"genre_scores_gemma":[0.7627188,0.00023954126,0.23088217,0.00013731503,0.00014871937,0.0002934081,0.0021202324,0.00037777715,0.0030819264],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9943705,0.0013344486,0.00047846537,0.0011315387,0.0023800703,0.00030501786],"domain_scores_gemma":[0.9659948,0.019305227,0.00435426,0.0046694903,0.0043750945,0.001301148],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004167564,0.0009706504,0.0011070215,0.014222747,0.0014907023,0.0022664238,0.0020207441,0.0019445793,0.0042413697],"category_scores_gemma":[0.038336087,0.00039036246,0.00086960936,0.007499421,0.0014072533,0.004036795,0.0030094679,0.0016616026,0.0013655598],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013314007,0.0008728829,0.27072677,0.0011394902,0.0008424676,0.0007797595,0.0016083423,0.042295944,0.044483438,0.081563294,0.012441875,0.54191434],"study_design_scores_gemma":[0.0001317845,0.0011988839,0.13794948,0.00019299818,0.00042664885,0.0034924648,0.0012287205,0.60587853,0.050838057,0.1777085,0.020591646,0.0003622609],"about_ca_topic_score_codex":0.0011114955,"about_ca_topic_score_gemma":0.0010735727,"teacher_disagreement_score":0.014222747,"about_ca_system_score_codex":0.0009141955,"about_ca_system_score_gemma":0.0011245318,"threshold_uncertainty_score":0.022040486},"labels":[],"label_agreement":null},{"id":"W3119029787","doi":"10.1007/s10664-020-09893-w","title":"Demystifying the challenges and benefits of analyzing user-reported logs in bug reports","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba; Concordia University","funders":"","keywords":"Debugging; Computer science; Software bug; Security bug; Process (computing); Software engineering; World Wide Web; Data science; Software; Programming language; Computer security","score_opus":0.03840174201774539,"score_gpt":0.26340921822509084,"score_spread":0.22500747620734546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3119029787","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.537558,0.007896166,0.39371836,0.0456034,0.0010669536,0.00044152993,0.0025623562,0.002878298,0.008274945],"genre_scores_gemma":[0.807378,0.0014863631,0.18632631,0.0014851927,0.00052392605,0.0001247818,0.0010355539,0.0003079371,0.0013320051],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9721272,0.01589537,0.0023932257,0.0017299721,0.007252817,0.0006014752],"domain_scores_gemma":[0.629755,0.25478646,0.02270295,0.0480988,0.042005878,0.002650988],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.047332253,0.0009916887,0.0010089321,0.008376349,0.0014000369,0.010108196,0.0020821106,0.0023506037,0.0009282849],"category_scores_gemma":[0.26471,0.0009336575,0.0007219394,0.005377555,0.0034144074,0.012972011,0.0036432736,0.004223877,0.000565781],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035945632,0.0005535735,0.47731426,0.0010920612,0.000574046,0.00048606755,0.011849592,0.008340313,0.011955501,0.02380886,0.010822519,0.45284382],"study_design_scores_gemma":[0.00012603463,0.00083051354,0.49452466,0.002574582,0.0006652569,0.0025930407,0.02302497,0.25117016,0.018024439,0.16217946,0.043702114,0.0005846846],"about_ca_topic_score_codex":0.011967186,"about_ca_topic_score_gemma":0.025155123,"teacher_disagreement_score":0.047332253,"about_ca_system_score_codex":0.0010187295,"about_ca_system_score_gemma":0.0042950893,"threshold_uncertainty_score":0.2503199},"labels":[],"label_agreement":null},{"id":"W3119686444","doi":"10.1007/s10664-021-10024-2","title":"A pragmatic approach for hyper-parameter tuning in search-based test case generation","year":2021,"lang":"en","type":"preprint","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Metric (unit); Computer science; Heuristic; Domain (mathematical analysis); Fine-tuning; Class (philosophy); Test case; Parameter space; Performance metric; Mathematical optimization; Machine learning; Artificial intelligence; Mathematics; Statistics; Engineering","score_opus":0.06445494654789755,"score_gpt":0.31047606952561757,"score_spread":0.24602112297772002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3119686444","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034694953,0.000067119894,0.9909554,0.00037937597,0.000036663565,0.00045340558,0.00005442647,0.0018271296,0.0027568927],"genre_scores_gemma":[0.16941792,0.00005001345,0.82634354,0.00064033817,0.00005225139,0.0012079576,0.00015243418,0.0007091593,0.0014264085],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.96605676,0.022323484,0.0018038033,0.0018997175,0.007101922,0.0008143376],"domain_scores_gemma":[0.9330552,0.04865988,0.0018684092,0.0092816595,0.006240278,0.0008945048],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01881284,0.0016079647,0.0018881515,0.0028395026,0.0015882319,0.004423411,0.004110392,0.004361264,0.011379559],"category_scores_gemma":[0.12222245,0.0017194066,0.0014232418,0.0016390199,0.002640289,0.0040041385,0.0067847827,0.0048427386,0.0027837881],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017060521,0.0011536578,0.0031279074,0.0010608375,0.00034103933,0.00073445635,0.0024626662,0.0882841,0.0521111,0.15341105,0.0129438685,0.68266314],"study_design_scores_gemma":[0.0006419237,0.0003391444,0.0009517655,0.00024970528,0.00015560945,0.00045076502,0.00034151482,0.8399852,0.017809669,0.12770575,0.011216202,0.00015281758],"about_ca_topic_score_codex":0.0011206875,"about_ca_topic_score_gemma":0.0022271548,"teacher_disagreement_score":0.01881284,"about_ca_system_score_codex":0.0012336834,"about_ca_system_score_gemma":0.0031860857,"threshold_uncertainty_score":0.09949297},"labels":[],"label_agreement":null},{"id":"W3119769340","doi":"10.1007/s10664-020-09917-5","title":"An exploratory study on the introduction and removal of different types of technical debt in deep learning frameworks","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Technical debt; Debt; Deep learning; Exploratory research; Quality (philosophy); Field (mathematics)","score_opus":0.018401164916278124,"score_gpt":0.2802219691050121,"score_spread":0.261820804188734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3119769340","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9960477,0.00009595014,0.0010377856,0.0001814843,0.000006828295,0.000047683625,0.00004161559,0.000019894236,0.002520988],"genre_scores_gemma":[0.9971117,0.00005574958,0.0015245653,0.00010707793,0.0000069518674,0.000037583985,0.000096541655,0.000018081884,0.0010418771],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99445146,0.0025624607,0.0003225613,0.000516574,0.0013703045,0.0007766506],"domain_scores_gemma":[0.88644075,0.08443769,0.013356247,0.0067346706,0.0052753664,0.003755264],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009577751,0.00026312514,0.00033526542,0.0009399501,0.0013848837,0.0020710505,0.0015099309,0.0015533358,0.0027383817],"category_scores_gemma":[0.079654366,0.0003401849,0.00028452635,0.0012093172,0.0015876883,0.0040442115,0.0018151572,0.0037374455,0.00028474536],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039005084,0.016867433,0.55453116,0.0012081633,0.00024266074,0.0020317764,0.082291596,0.0058345203,0.021201082,0.036476538,0.0048302435,0.27058432],"study_design_scores_gemma":[0.00042504008,0.0066607753,0.78640085,0.0008235967,0.00030192963,0.0014041702,0.09247161,0.029159464,0.018194132,0.018531803,0.045395605,0.00023103088],"about_ca_topic_score_codex":0.0029826972,"about_ca_topic_score_gemma":0.005832723,"teacher_disagreement_score":0.009577751,"about_ca_system_score_codex":0.0019435274,"about_ca_system_score_gemma":0.0019391119,"threshold_uncertainty_score":0.050652623},"labels":[],"label_agreement":null},{"id":"W3121596715","doi":"10.1007/s10664-017-9521-5","title":"Do developers update their library dependencies?","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":335,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Japan Society for the Promotion of Science","keywords":"Reuse; Dependency (UML); Computer science; Workload; Exploit; Software; World Wide Web; Software engineering; Data science; Computer security; Engineering","score_opus":0.03368879006535396,"score_gpt":0.27991691801105534,"score_spread":0.24622812794570137,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3121596715","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9548504,0.00075056386,0.005463851,0.005922866,0.00011510686,0.00006150495,0.0012215924,0.0006858766,0.030928278],"genre_scores_gemma":[0.98866165,0.00036631676,0.0021991646,0.00088197464,0.000053067815,0.00003228203,0.0006989458,0.00033027717,0.0067763985],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9903328,0.002700974,0.0008634981,0.0012044355,0.003983513,0.0009148067],"domain_scores_gemma":[0.6781005,0.17249158,0.074234806,0.034428388,0.034123104,0.0066215578],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.010384428,0.0004406657,0.0004082184,0.0037461696,0.0013201408,0.0035994942,0.0016320902,0.0019040348,0.0099123735],"category_scores_gemma":[0.21997064,0.0008969265,0.00032431528,0.0033342899,0.0014413485,0.009339746,0.0017561277,0.0022612917,0.0023015453],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028912505,0.00048653272,0.8194011,0.00021580106,0.00011984352,0.0005933955,0.011374808,0.00057392305,0.0014679356,0.0053179683,0.013480783,0.14667885],"study_design_scores_gemma":[0.000105552186,0.00026019613,0.9167011,0.00042440605,0.00034959117,0.0016638163,0.016427245,0.004835694,0.0052137836,0.012238606,0.041661892,0.00011804588],"about_ca_topic_score_codex":0.014999587,"about_ca_topic_score_gemma":0.02827103,"teacher_disagreement_score":0.98961556,"about_ca_system_score_codex":0.0019584303,"about_ca_system_score_gemma":0.0038065747,"threshold_uncertainty_score":0.054918766},"labels":[],"label_agreement":null},{"id":"W3123086775","doi":"10.1007/s10664-021-09982-4","title":"Can Offline Testing of Deep Neural Networks Replace Their Online Testing?","year":2021,"lang":"en","type":"preprint","venue":"Empirical Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"H2020 European Research Council; Canada Research Chairs; Ministry of Education; National Research Foundation of Korea; Fonds National de la Recherche Luxembourg; National Research Foundation; Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Computer science; Online and offline; Test strategy; Manual testing; Orthogonal array testing; White-box testing; Context (archaeology); Black-box testing; Exploit; Model-based testing; Machine learning; Artificial intelligence; Test case; Computer security; Operating system; Software","score_opus":0.03600581497936919,"score_gpt":0.27307237304567505,"score_spread":0.23706655806630586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3123086775","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41725865,0.0018078111,0.55886406,0.0053563616,0.00056393276,0.0002164974,0.00047823938,0.0042174244,0.011236999],"genre_scores_gemma":[0.958581,0.00013075596,0.039564747,0.0004852441,0.00005112372,0.00008724066,0.00016384744,0.0001994102,0.00073673716],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98641044,0.008255851,0.0006476193,0.0016727254,0.002268869,0.00074447755],"domain_scores_gemma":[0.86898315,0.097686395,0.009100891,0.017312737,0.005687712,0.0012290532],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014047326,0.0016807185,0.0008862258,0.0009884975,0.0003751558,0.0014636257,0.0034358697,0.0022614594,0.0035593451],"category_scores_gemma":[0.11362766,0.00060932036,0.00056471507,0.00067682937,0.0027961896,0.0061310246,0.0021647236,0.003090594,0.0009798696],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019216975,0.0013249088,0.072148085,0.00062324095,0.00026857923,0.0007425373,0.00039085574,0.42325607,0.015367592,0.029810693,0.008691807,0.44545385],"study_design_scores_gemma":[0.000055087596,0.00040445584,0.0041400087,0.0001214575,0.000031836164,0.00017731995,0.000091246984,0.9542134,0.0113047995,0.027800536,0.0016261331,0.000033802364],"about_ca_topic_score_codex":0.0031166319,"about_ca_topic_score_gemma":0.0028387008,"teacher_disagreement_score":0.014047326,"about_ca_system_score_codex":0.0014539729,"about_ca_system_score_gemma":0.0017373138,"threshold_uncertainty_score":0.074290216},"labels":[],"label_agreement":null},{"id":"W3123088576","doi":"10.1007/s10664-020-09900-0","title":"Investigating design anti-pattern and design pattern mutations and their change- and fault-proneness","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Computer Research Institute of Montréal; Polytechnique Montréal","funders":"","keywords":"Software design pattern; Structural pattern; Software design; Computer science; Design pattern; Software evolution; Software quality; Software; Software development; Software engineering; Software construction; Programming language","score_opus":0.09046305438679865,"score_gpt":0.2872581248634783,"score_spread":0.19679507047667966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3123088576","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9973074,0.000071233495,0.0020202235,0.000050888575,0.0000022507375,0.00001320583,0.00006699455,0.000021993048,0.0004457995],"genre_scores_gemma":[0.99873966,0.000017324997,0.0010091352,0.000007415255,0.0000013618836,0.0000087193175,0.000059410053,0.0000069467155,0.00015016996],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9945041,0.002355385,0.0004960385,0.0009199658,0.0014572084,0.00026723056],"domain_scores_gemma":[0.75764173,0.1747767,0.04703725,0.012071723,0.0071450467,0.0013275591],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067707193,0.0003487314,0.00026780204,0.0018849864,0.0002778272,0.0009966785,0.000857642,0.0007969543,0.0019624934],"category_scores_gemma":[0.11432311,0.000292898,0.00041885776,0.0014342013,0.0008645757,0.0017827207,0.0007298924,0.0011461513,0.00019506425],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044379753,0.0004348337,0.9511425,0.00007754081,0.00023275215,0.00021592999,0.001185961,0.0051816925,0.00511491,0.0017062429,0.00013457665,0.03412922],"study_design_scores_gemma":[0.00005978881,0.00073955965,0.93201226,0.00003866604,0.0001976011,0.0008660775,0.001786676,0.051741946,0.0064802305,0.0053383266,0.000703149,0.000035716414],"about_ca_topic_score_codex":0.0011074498,"about_ca_topic_score_gemma":0.0019765839,"teacher_disagreement_score":0.0067707193,"about_ca_system_score_codex":0.00060659985,"about_ca_system_score_gemma":0.00066280446,"threshold_uncertainty_score":0.03580737},"labels":[],"label_agreement":null},{"id":"W3129150043","doi":"10.1007/s10664-020-09902-y","title":"On the Removal of Feature Toggles","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University","funders":"","keywords":"Python (programming language); Feature (linguistics); Computer science; Source code; Cyclomatic complexity; Software; Data science; Software engineering; Programming language","score_opus":0.024005836158592978,"score_gpt":0.27051804039179844,"score_spread":0.24651220423320547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3129150043","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26561174,0.0020523726,0.65127873,0.0123526985,0.0014758633,0.0002657575,0.000935379,0.006134538,0.05989295],"genre_scores_gemma":[0.7237977,0.00082405447,0.24125686,0.0018552797,0.00054013444,0.00010161232,0.0015153444,0.001954973,0.02815414],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99486214,0.0016556684,0.00020357058,0.0006421348,0.0019867592,0.0006498408],"domain_scores_gemma":[0.94535077,0.026728928,0.0024236925,0.018890647,0.005873825,0.00073203276],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005107956,0.0007893287,0.0011921955,0.0022007392,0.0018562141,0.0026756094,0.0027292583,0.0024565568,0.009828616],"category_scores_gemma":[0.059493095,0.0005290792,0.0010958998,0.0020065983,0.0021852176,0.005316001,0.0030775152,0.0023810987,0.003163823],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010431948,0.00059788435,0.013332577,0.00034198124,0.00014879623,0.00070127245,0.0006635691,0.017035166,0.013023469,0.07428856,0.030526733,0.84829676],"study_design_scores_gemma":[0.0004386247,0.000790369,0.052335396,0.00053235324,0.0006387322,0.0023721054,0.0022817615,0.41372076,0.04836959,0.37668535,0.10156377,0.0002711979],"about_ca_topic_score_codex":0.0059734234,"about_ca_topic_score_gemma":0.009671143,"teacher_disagreement_score":0.009828616,"about_ca_system_score_codex":0.0006545856,"about_ca_system_score_gemma":0.002315817,"threshold_uncertainty_score":0.03288001},"labels":[],"label_agreement":null},{"id":"W3130317758","doi":"10.1007/s10664-022-10170-1","title":"Optimal priority assignment for real-time systems: a coevolution-based approach","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Real-Time Systems Scheduling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"European Research Council; Natural Sciences and Engineering Research Council of Canada; European Commission; Université du Luxembourg","keywords":"Coevolution; Computer science; Real-time computing; Biology","score_opus":0.01832058550148069,"score_gpt":0.24808763269531386,"score_spread":0.22976704719383317,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3130317758","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03855381,0.00048167177,0.95646745,0.00032248508,0.000042857613,0.000099401674,0.000024392948,0.00020004192,0.0038078744],"genre_scores_gemma":[0.62309957,0.0003512991,0.37270427,0.00022506439,0.00006716923,0.00029561142,0.000093025,0.0001299922,0.0030340233],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99925,0.00027797735,0.000037851,0.00014210332,0.0001974818,0.00009448084],"domain_scores_gemma":[0.9975477,0.0015751962,0.00020143538,0.00012414559,0.0004174662,0.00013397736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021788056,0.001145737,0.001225058,0.0018356644,0.0006790076,0.0012156003,0.0019979437,0.0015584581,0.0023316194],"category_scores_gemma":[0.006256476,0.00066174666,0.00088488875,0.0009772638,0.0011371049,0.0012043478,0.001119359,0.0014387469,0.00024948],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002787393,0.000062400264,0.0008081261,0.00004819427,0.00004940666,0.000052961797,0.00008312893,0.9612432,0.0010457869,0.010712361,0.0004259801,0.025440583],"study_design_scores_gemma":[0.0000059450754,0.000011759592,0.000065918575,0.000003415435,0.000005282629,0.000008737861,0.000009941497,0.9971928,0.00012325564,0.0023878103,0.00018227332,0.0000028466627],"about_ca_topic_score_codex":0.007372376,"about_ca_topic_score_gemma":0.005395437,"teacher_disagreement_score":0.007372376,"about_ca_system_score_codex":0.0017498903,"about_ca_system_score_gemma":0.0017799769,"threshold_uncertainty_score":0.014658928},"labels":[],"label_agreement":null},{"id":"W3134065202","doi":"10.1007/s10664-020-09904-w","title":"Helping or not helping? Why and how trivial packages impact the npm ecosystem","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Publication; Computer science; JavaScript; Documentation; World Wide Web; Overhead (engineering); Publishing; Dependency (UML); Software engineering; Operating system; Business; Advertising","score_opus":0.03123396730809244,"score_gpt":0.2809393432019691,"score_spread":0.24970537589387667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3134065202","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8406235,0.00081188243,0.007014139,0.014761865,0.00012730116,0.00007247787,0.0003823656,0.00025369792,0.13595276],"genre_scores_gemma":[0.99428564,0.00029519972,0.0017377762,0.00077560043,0.000034170578,0.000024709205,0.0000968709,0.00010203425,0.0026480935],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99130696,0.004539384,0.0001865813,0.00072008563,0.002129311,0.0011176232],"domain_scores_gemma":[0.94587225,0.031900477,0.0061769406,0.006049593,0.0064826203,0.0035181488],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.009651544,0.00038913544,0.00046300018,0.0019456334,0.0025220287,0.0057285246,0.0013329954,0.0020972716,0.023543706],"category_scores_gemma":[0.070722446,0.00033936213,0.0004169893,0.0025412259,0.00580538,0.014350945,0.005537769,0.0024953838,0.0027737578],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076406606,0.0017552364,0.39149725,0.00070521195,0.00019459105,0.0007634229,0.026559781,0.002412663,0.004716223,0.27601978,0.02965317,0.2649586],"study_design_scores_gemma":[0.00015415259,0.0006751472,0.44506258,0.0010389559,0.00027588886,0.000915783,0.05963257,0.013590078,0.0035940285,0.392404,0.08250455,0.00015225346],"about_ca_topic_score_codex":0.0077010677,"about_ca_topic_score_gemma":0.01036572,"teacher_disagreement_score":0.99034846,"about_ca_system_score_codex":0.0029174222,"about_ca_system_score_gemma":0.0031957256,"threshold_uncertainty_score":0.07876158},"labels":[],"label_agreement":null},{"id":"W3135032580","doi":"10.1007/s10664-020-09913-9","title":"Software product-line evaluation in the large","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Saab; Vetenskapsrådet; Deutsche Forschungsgemeinschaft; Deutscher Akademischer Austauschdienst","keywords":"Domain engineering; Software product line; Software engineering; Product (mathematics); Domain (mathematical analysis); New product development; Software development; Product engineering; Systems engineering; Computer science; Process management; Software; Engineering; Risk analysis (engineering); Product design; Software construction; Business; Marketing","score_opus":0.07412629567025701,"score_gpt":0.34508034406743116,"score_spread":0.27095404839717413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3135032580","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9191659,0.0009155923,0.050738648,0.0016616258,0.00006541886,0.0011468531,0.00029986905,0.00034771458,0.025658421],"genre_scores_gemma":[0.9804115,0.00014102327,0.017405424,0.000107156106,0.000017522221,0.00040229879,0.00024191634,0.000060431288,0.0012126305],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9402511,0.04494079,0.0017434261,0.0023494519,0.009648953,0.001066352],"domain_scores_gemma":[0.7964751,0.14038666,0.013447665,0.008211459,0.03693274,0.004546339],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05003994,0.00059485517,0.0005210812,0.0041480632,0.0014291981,0.0037244747,0.0012301321,0.0008598697,0.004627007],"category_scores_gemma":[0.11924969,0.00026324386,0.00038466262,0.0027180742,0.001894926,0.006021253,0.0031557367,0.0010905328,0.0008180911],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00153377,0.0026266563,0.14952476,0.001323317,0.00019825954,0.0014184689,0.06965172,0.010910132,0.009703649,0.020388128,0.01336258,0.71935844],"study_design_scores_gemma":[0.00039986987,0.008586373,0.5224539,0.0027711138,0.00024890882,0.0012089624,0.12212211,0.11580098,0.023514,0.046266202,0.15602705,0.00060060504],"about_ca_topic_score_codex":0.0029998673,"about_ca_topic_score_gemma":0.002635073,"teacher_disagreement_score":0.05003994,"about_ca_system_score_codex":0.0047416426,"about_ca_system_score_gemma":0.0024967967,"threshold_uncertainty_score":0.26463962},"labels":[],"label_agreement":null},{"id":"W3135374914","doi":"10.1007/s10664-020-09892-x","title":"variED: an editor for collaborative, real-time feature modeling","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Deutsche Forschungsgemeinschaft; Deutscher Akademischer Austauschdienst","keywords":"Computer science; Feature model; Usability; Feature (linguistics); Software engineering; Software product line; Merge (version control); Software; Artifact (error); Human–computer interaction; Software development; Artificial intelligence; Information retrieval; Programming language","score_opus":0.040571527766392786,"score_gpt":0.3121451311731008,"score_spread":0.271573603406708,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3135374914","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012455336,0.00016812187,0.916764,0.000308878,0.0007547702,0.00021465968,0.0016487538,0.073277995,0.0056173643],"genre_scores_gemma":[0.038530156,0.00046010432,0.89830595,0.00044533084,0.0005809761,0.00061501976,0.007430935,0.034924597,0.018706998],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9940942,0.00148372,0.0008542711,0.0011348824,0.0021906036,0.00024240451],"domain_scores_gemma":[0.9777034,0.012427258,0.0008413993,0.0058319196,0.0021931378,0.0010028642],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0074286815,0.0020661524,0.0016827447,0.0028403602,0.0012768168,0.0052155443,0.0071502733,0.0030548964,0.052885465],"category_scores_gemma":[0.027231598,0.0021154024,0.002933522,0.0017839863,0.0016333121,0.007752877,0.008039049,0.0059728203,0.021045888],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013487185,0.0005687317,0.0019775096,0.0022963667,0.00028620733,0.0040102885,0.0026577138,0.032177366,0.026053084,0.17677756,0.33198756,0.4198589],"study_design_scores_gemma":[0.00029539163,0.00016989384,0.00029614067,0.0003576515,0.000105919426,0.0012778174,0.00017141046,0.12480733,0.014030319,0.043590873,0.8146958,0.00020144151],"about_ca_topic_score_codex":0.00072813436,"about_ca_topic_score_gemma":0.0015030847,"teacher_disagreement_score":0.052885465,"about_ca_system_score_codex":0.0010567423,"about_ca_system_score_gemma":0.0021392542,"threshold_uncertainty_score":0.17691952},"labels":[],"label_agreement":null},{"id":"W3135659312","doi":"10.1007/s10664-020-09918-4","title":"Wikifying software artifacts","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Wikis in Education and Collaboration","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Leverage (statistics); Software; Precision and recall; Recall; Domain (mathematical analysis); Data mining; Artificial intelligence; Machine learning; Information retrieval; Software engineering; Programming language","score_opus":0.03677555461118149,"score_gpt":0.34255519330012735,"score_spread":0.30577963868894587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3135659312","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.78141844,0.00093761476,0.05728669,0.0014592357,0.00018665688,0.00058248924,0.00068318326,0.0007713808,0.15667433],"genre_scores_gemma":[0.9613684,0.00039285008,0.020317046,0.00009721172,0.00002669595,0.00012085013,0.00076438335,0.00011252145,0.01680014],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98603255,0.0068060635,0.0011733184,0.0013748233,0.0039380747,0.0006751495],"domain_scores_gemma":[0.9293242,0.032461073,0.00679569,0.019660193,0.010311736,0.0014470486],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.009123417,0.000716805,0.0003323472,0.006045232,0.0022941574,0.008579274,0.0011402782,0.0012335259,0.006228023],"category_scores_gemma":[0.057149645,0.00048564514,0.00045908007,0.0045758686,0.0026864621,0.0075600343,0.003860711,0.0012689299,0.0016548502],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033276438,0.0011434691,0.17498724,0.0010952521,0.00015453213,0.0010383255,0.052105904,0.002585362,0.010416636,0.22985879,0.009850123,0.51643157],"study_design_scores_gemma":[0.00013472477,0.0010363888,0.28458273,0.0020440302,0.00047090545,0.00382662,0.120788604,0.022578191,0.08067915,0.147487,0.33615786,0.00021389345],"about_ca_topic_score_codex":0.0044496804,"about_ca_topic_score_gemma":0.005602787,"teacher_disagreement_score":0.99087656,"about_ca_system_score_codex":0.0021635867,"about_ca_system_score_gemma":0.004412224,"threshold_uncertainty_score":0.04824978},"labels":[],"label_agreement":null},{"id":"W3137668562","doi":"10.1007/s10664-020-09929-1","title":"Release synchronization in software ecosystems","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Ecosystem; Context (archaeology); Synchronization (alternating current); Project team; Computer science; Knowledge management; Ecology; Telecommunications; Geography","score_opus":0.015028767916721806,"score_gpt":0.25037503824282314,"score_spread":0.23534627032610134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3137668562","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96872395,0.00040229643,0.022315962,0.00068229274,0.00002901599,0.000017975866,0.000048264174,0.00016416713,0.007616035],"genre_scores_gemma":[0.9983132,0.000058437476,0.0009117208,0.000016432052,0.000013308541,0.000007862913,0.000020966467,0.000015915362,0.00064218603],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.99723864,0.0013580011,0.00019061138,0.00048186915,0.00044141733,0.0002894565],"domain_scores_gemma":[0.92915064,0.05060212,0.011238723,0.004761998,0.00229939,0.0019471466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070975707,0.0002615312,0.00040742627,0.0017036493,0.0008304105,0.002774709,0.0007873343,0.001127783,0.0045936084],"category_scores_gemma":[0.06569913,0.00039544114,0.0003599905,0.0013411843,0.0017583962,0.004693478,0.0017121019,0.0011708947,0.00044866744],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016455454,0.0010320201,0.34006232,0.00035575926,0.0003242951,0.0012959582,0.005313831,0.101431146,0.011168518,0.37593415,0.0044581885,0.1569783],"study_design_scores_gemma":[0.00027036096,0.0007105622,0.21324886,0.000148942,0.00026374945,0.0008356204,0.0054196124,0.3843637,0.0048656925,0.38484266,0.004912312,0.00011798361],"about_ca_topic_score_codex":0.002295223,"about_ca_topic_score_gemma":0.0013504499,"teacher_disagreement_score":0.0070975707,"about_ca_system_score_codex":0.0011121024,"about_ca_system_score_gemma":0.0010480144,"threshold_uncertainty_score":0.037536025},"labels":[],"label_agreement":null},{"id":"W3139738689","doi":"10.1007/s10664-021-10028-y","title":"An exploratory study on the repeatedly shared external links on Stack Overflow","year":2021,"lang":"en","type":"preprint","venue":"Empirical Software Engineering","topic":"Expert finding and Q&A systems","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); Queen's University","funders":"","keywords":"Computer science; Stack (abstract data type); Point (geometry); World Wide Web; Operating system","score_opus":0.05440002308055616,"score_gpt":0.29818777125471374,"score_spread":0.24378774817415758,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3139738689","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9978162,0.000031316125,0.0003500066,0.00008429653,0.000004272179,0.00002373006,0.000034885856,0.000011056319,0.0016442116],"genre_scores_gemma":[0.99840754,0.00004814483,0.00042965173,0.000053270887,0.000011225882,0.000019657142,0.000065006214,0.000019180885,0.0009463957],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9972696,0.001430845,0.00015106777,0.00022391404,0.0005897077,0.00033488244],"domain_scores_gemma":[0.92543906,0.059465405,0.0073172925,0.0029235177,0.0026980022,0.0021567852],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0037021542,0.00033846422,0.0003227059,0.002821652,0.0035158677,0.0024121373,0.0013388003,0.0020202894,0.0040667257],"category_scores_gemma":[0.063957274,0.00043936793,0.00019679284,0.0024524971,0.002106978,0.00448424,0.0034938445,0.0017625885,0.00046322442],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011306555,0.0027480107,0.4762836,0.00042172466,0.00010846858,0.008248931,0.41195026,0.00088209443,0.010918328,0.0059039923,0.0019535755,0.07945034],"study_design_scores_gemma":[0.00008512358,0.0018232093,0.60168445,0.00033903364,0.00013371528,0.004073703,0.36152735,0.0034794006,0.0062109507,0.0052064313,0.015292621,0.00014404696],"about_ca_topic_score_codex":0.004292783,"about_ca_topic_score_gemma":0.007823595,"teacher_disagreement_score":0.9971784,"about_ca_system_score_codex":0.00087668764,"about_ca_system_score_gemma":0.0011978409,"threshold_uncertainty_score":0.019579053},"labels":[],"label_agreement":null},{"id":"W3147362533","doi":"10.1007/s10664-021-09944-w","title":"Revisiting the VCCFinder approach for the identification of vulnerability-contributing commits","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"Fonds National de la Recherche Luxembourg; European Commission","keywords":"Commit; Computer science; Identification (biology); Vulnerability (computing); Replicate; Artificial intelligence; Machine learning; Replication (statistics); Software; Set (abstract data type); Software deployment; Data science; Software engineering; Computer security; Programming language; Database","score_opus":0.03679435711364231,"score_gpt":0.3062734892830032,"score_spread":0.26947913216936087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3147362533","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16784161,0.00087287545,0.8099133,0.0017162459,0.00027042456,0.00056362874,0.0022960694,0.01237735,0.0041485312],"genre_scores_gemma":[0.56664884,0.00014117014,0.4250371,0.00037548452,0.00014973637,0.00031258998,0.003972861,0.00040731585,0.00295492],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9895902,0.003704948,0.00072527974,0.0025834679,0.0028616078,0.00053454813],"domain_scores_gemma":[0.91817474,0.045517813,0.007017333,0.014107426,0.013736703,0.001445949],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011131925,0.0009017745,0.0011822212,0.0079924315,0.0013631061,0.002496773,0.0032541999,0.002480243,0.002007028],"category_scores_gemma":[0.05187663,0.00047837864,0.0009182373,0.0025850858,0.0019034467,0.0038057303,0.0033074599,0.0032979583,0.0013079809],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006035688,0.0011630676,0.112592556,0.00079081394,0.00027608764,0.0007455966,0.0018534225,0.08489836,0.013624346,0.015403167,0.026185578,0.7418635],"study_design_scores_gemma":[0.000036361995,0.00019764443,0.010038457,0.00016976542,0.00004410754,0.0005205762,0.00052113185,0.95103824,0.0116633745,0.018160561,0.0075403643,0.0000694746],"about_ca_topic_score_codex":0.004763141,"about_ca_topic_score_gemma":0.007726825,"teacher_disagreement_score":0.011131925,"about_ca_system_score_codex":0.0012320312,"about_ca_system_score_gemma":0.0031308061,"threshold_uncertainty_score":0.058871984},"labels":[],"label_agreement":null},{"id":"W3152918650","doi":"10.1007/s10664-023-10314-x","title":"Evaluating pre-trained models for user feedback analysis in software engineering: a study on classification of app-reviews","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Machine learning; Artificial intelligence; F1 score; Macro; Set (abstract data type); Task (project management); Binary classification; Support vector machine; Data mining; Engineering","score_opus":0.15107932163450608,"score_gpt":0.40137155726348567,"score_spread":0.2502922356289796,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3152918650","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9693755,0.0017337397,0.024027567,0.00028633358,0.0001826134,0.00026617874,0.00088440033,0.0015513311,0.0016923426],"genre_scores_gemma":[0.9757111,0.00029312042,0.018400788,0.0001234232,0.000066437264,0.00015067695,0.0032201395,0.00013873282,0.0018956979],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9910953,0.004404254,0.00080830796,0.0018732453,0.0014177454,0.00040098626],"domain_scores_gemma":[0.8748576,0.103799924,0.0030447373,0.005029435,0.011851862,0.0014164327],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.011124551,0.0015995802,0.0014535597,0.0025069974,0.00084640924,0.0023439897,0.0021041897,0.0023733014,0.0009961292],"category_scores_gemma":[0.054604802,0.0005244246,0.0011359736,0.0014390876,0.0005410224,0.0028307673,0.0011602737,0.0024410402,0.0013794728],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0055603716,0.007036302,0.24858257,0.0014760012,0.0014684917,0.00044318187,0.0030640182,0.085010566,0.011288842,0.00084322965,0.018163668,0.61706275],"study_design_scores_gemma":[0.00013180343,0.0018336348,0.058121856,0.00017957963,0.0004365423,0.0003125734,0.00070439413,0.9259763,0.0089081945,0.0008217448,0.0024829125,0.000090574635],"about_ca_topic_score_codex":0.012213194,"about_ca_topic_score_gemma":0.013390184,"teacher_disagreement_score":0.98887545,"about_ca_system_score_codex":0.00222015,"about_ca_system_score_gemma":0.0020111306,"threshold_uncertainty_score":0.058832943},"labels":[],"label_agreement":null},{"id":"W3157341372","doi":"10.1007/s10664-021-09977-1","title":"Individual differences limit predicting well-being and productivity using software repositories: a longitudinal industrial study","year":2021,"lang":"en","type":"preprint","venue":"Empirical Software Engineering","topic":"Mental Health Research Topics","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Oulun Yliopisto; Academy of Finland; KAUTE-Säätiö","keywords":"Productivity; Commit; Software; Computer science; Variance (accounting); Software development; Data science; Software metric; Software engineering; Work (physics); Software sizing; Personal software process; Software quality; Software construction; Engineering; Database; Business","score_opus":0.1754318038953424,"score_gpt":0.39816209944449493,"score_spread":0.22273029554915252,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3157341372","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9995522,0.000027635531,0.00016347774,0.000027558364,0.0000020883374,0.000009730448,0.00004764216,0.0000016462593,0.00016786056],"genre_scores_gemma":[0.9996125,0.000019109291,0.00012331351,0.000011936708,0.0000027810363,0.000015680154,0.00006709217,0.0000020286936,0.00014562636],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99715084,0.0015226157,0.00013627639,0.000432878,0.00041512752,0.00034238602],"domain_scores_gemma":[0.98605496,0.0064546173,0.002503129,0.0015122505,0.0016630157,0.0018119527],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0069590947,0.0002842387,0.00037157774,0.0012052297,0.00087950897,0.0014500438,0.0005313729,0.00077458494,0.0014129519],"category_scores_gemma":[0.017894687,0.00039357576,0.00051449594,0.00094377523,0.0007258931,0.0010795606,0.0010492072,0.0013601306,0.00050745805],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000056738063,0.0004706836,0.99447554,0.0000057031307,0.000029993475,0.00006418942,0.0019003948,0.00007081188,0.000105421685,0.00004590308,0.000100998965,0.0026736807],"study_design_scores_gemma":[0.000004394023,0.00026508007,0.9965688,0.000011886309,0.000018210236,0.00006899251,0.001968693,0.0006703706,0.00008847027,0.00010797397,0.00021720631,0.000010045893],"about_ca_topic_score_codex":0.0060709324,"about_ca_topic_score_gemma":0.0067372825,"teacher_disagreement_score":0.0069590947,"about_ca_system_score_codex":0.0004483842,"about_ca_system_score_gemma":0.00056692294,"threshold_uncertainty_score":0.036803663},"labels":[],"label_agreement":null},{"id":"W3157562347","doi":"10.1007/s10664-020-09926-4","title":"The nature of build changes","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Dependency (UML); Java; Plug-in; Software evolution; Software maintenance; Software engineering; Software; Legacy system; Software system; Programming language; Software construction","score_opus":0.015523738207438768,"score_gpt":0.28122853624808597,"score_spread":0.2657047980406472,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3157562347","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92870617,0.004487102,0.044640798,0.0005996328,0.00041706327,0.00037563432,0.006436094,0.004360026,0.009977406],"genre_scores_gemma":[0.9500392,0.0012953589,0.031373445,0.00027144846,0.00014843738,0.00025838756,0.011236277,0.001002339,0.004375127],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9893287,0.0014855823,0.0010953286,0.0023625987,0.0052690743,0.0004587083],"domain_scores_gemma":[0.91399974,0.044053458,0.013724881,0.010989114,0.01615919,0.0010735363],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041509187,0.0006121863,0.0004947178,0.007766322,0.0009189095,0.0024457113,0.0009802697,0.00082215597,0.0011469126],"category_scores_gemma":[0.052033357,0.00060680055,0.0006578013,0.003947315,0.00062824966,0.0030186165,0.0015568562,0.0011588394,0.00089960935],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004380662,0.00012512384,0.52808917,0.0012977221,0.0003291909,0.0015328777,0.0066593457,0.0047991956,0.01937775,0.0021197838,0.012144219,0.4230875],"study_design_scores_gemma":[0.000023938748,0.00025796858,0.86913306,0.00062502356,0.00028214743,0.0029498243,0.0025609327,0.021845143,0.022822864,0.0027936993,0.076543555,0.00016186271],"about_ca_topic_score_codex":0.0031490403,"about_ca_topic_score_gemma":0.0053048353,"teacher_disagreement_score":0.007766322,"about_ca_system_score_codex":0.0008036872,"about_ca_system_score_gemma":0.00063956284,"threshold_uncertainty_score":0.02195239},"labels":[],"label_agreement":null},{"id":"W3159076496","doi":"10.1007/s10664-020-09910-y","title":"Promises and challenges of microservices: an exploratory study","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":95,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Microservices; Computer science; Software deployment; Scalability; Reuse; Best practice; Software development; Process (computing); Software engineering; Code reuse; Software; Knowledge management; Engineering; Cloud computing; Database; Management","score_opus":0.04185144518438682,"score_gpt":0.2711256357909792,"score_spread":0.22927419060659238,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3159076496","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98637867,0.00024400021,0.00069848215,0.0011775799,0.000006108991,0.000056905763,0.000050519146,0.00000862753,0.011379107],"genre_scores_gemma":[0.9987263,0.00015687343,0.00028794038,0.00006552373,0.0000068988634,0.000025371915,0.000021740198,0.0000050600593,0.0007042467],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9953328,0.002596754,0.00014410721,0.00026385675,0.0011585553,0.00050386414],"domain_scores_gemma":[0.8690165,0.10909955,0.009849322,0.002703977,0.005428968,0.0039016071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009827132,0.00020994413,0.00021838653,0.0012181957,0.002481442,0.0033571287,0.0012477029,0.0011919071,0.0064037573],"category_scores_gemma":[0.050783854,0.0003379662,0.00021577057,0.0015625911,0.0028129378,0.008007493,0.0022078645,0.0021653303,0.0006014331],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022615013,0.0054445993,0.550806,0.001097283,0.000081879705,0.0024023352,0.17168346,0.0023817471,0.0043227645,0.12864316,0.0048056673,0.12606964],"study_design_scores_gemma":[0.0001031497,0.003138684,0.36683983,0.0007035791,0.0000965827,0.0012201067,0.5562902,0.008558925,0.0035095117,0.021816839,0.037634455,0.00008810324],"about_ca_topic_score_codex":0.0020428114,"about_ca_topic_score_gemma":0.0037866912,"teacher_disagreement_score":0.009827132,"about_ca_system_score_codex":0.0018177103,"about_ca_system_score_gemma":0.002431742,"threshold_uncertainty_score":0.051971495},"labels":[],"label_agreement":null},{"id":"W3163691159","doi":"10.1007/s10664-021-09954-8","title":"On using Stack Overflow comment-edit pairs to recommend code maintenance changes","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Commit; Code (set theory); Source code; Code review; Code smell; Data mining; Programming language; Point (geometry); Interface (matter); Static program analysis; Set (abstract data type); Software; Database; Software development; Software quality; Operating system","score_opus":0.05172537609133042,"score_gpt":0.3103960695023846,"score_spread":0.25867069341105414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3163691159","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6899592,0.0023995775,0.2691909,0.0035326476,0.0007645626,0.0009781377,0.0037086813,0.0146079855,0.014858222],"genre_scores_gemma":[0.82551813,0.00039953622,0.16050836,0.0008274178,0.00025990387,0.00017710842,0.0046341782,0.00031234868,0.007362953],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99739516,0.0008049625,0.00020820869,0.0005193134,0.00087826484,0.00019404183],"domain_scores_gemma":[0.9648567,0.026067542,0.0009411076,0.0018713893,0.005596375,0.00066681154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042079166,0.0013161892,0.0013134367,0.004612102,0.0014779993,0.0022057425,0.0019275398,0.0032296313,0.0057501565],"category_scores_gemma":[0.035258695,0.0005875561,0.0006272203,0.0020291284,0.00050177414,0.0041373144,0.0014917126,0.0018164009,0.0025999425],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038211448,0.0016281037,0.06542669,0.00035813978,0.00036105196,0.0002430553,0.0007034273,0.023196135,0.013257322,0.0023673028,0.049761556,0.8388761],"study_design_scores_gemma":[0.00042605025,0.00079648755,0.02809656,0.00010883169,0.00024355961,0.00026712572,0.0009078412,0.9459252,0.010749635,0.005375653,0.0069768997,0.00012633442],"about_ca_topic_score_codex":0.03999728,"about_ca_topic_score_gemma":0.069588736,"teacher_disagreement_score":0.03999728,"about_ca_system_score_codex":0.0009600267,"about_ca_system_score_gemma":0.0019618296,"threshold_uncertainty_score":0.07952893},"labels":[],"label_agreement":null},{"id":"W3166692829","doi":"10.1007/s10664-021-09979-z","title":"Studying backers and hunters in bounty issue addressing process of open source projects","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Manitoba; Concordia University; Huawei Technologies (Canada)","funders":"","keywords":"Value (mathematics); Process (computing); Business; Computer science","score_opus":0.05329574829042888,"score_gpt":0.33616590050921935,"score_spread":0.28287015221879047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3166692829","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9913862,0.000081033024,0.0007514813,0.00029965115,0.0000150259275,0.000021515889,0.000015404717,0.000007921688,0.0074218963],"genre_scores_gemma":[0.99494284,0.00006499195,0.0004703985,0.000056246863,0.000011210583,0.000016092818,0.000027338856,0.00001030635,0.0044004577],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99734336,0.0011534719,0.000103444334,0.0003169792,0.0005913474,0.0004913493],"domain_scores_gemma":[0.9674442,0.018802162,0.006882548,0.0012742468,0.0022407835,0.003356054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004041289,0.0001982186,0.00025548146,0.0021212255,0.0035751062,0.0032781684,0.0007940374,0.001442473,0.005958419],"category_scores_gemma":[0.03408736,0.00026888476,0.00020680227,0.0013154035,0.0018898417,0.0038324744,0.0028607205,0.0017051224,0.00057771447],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004048898,0.00074357155,0.62087196,0.00015730687,0.000046741512,0.0013231712,0.28692663,0.00032055285,0.0023947156,0.020822417,0.0038776933,0.062110312],"study_design_scores_gemma":[0.000022147646,0.00028211396,0.5781625,0.00017258846,0.000041261283,0.00064341305,0.3904479,0.0026243979,0.0008924786,0.006818934,0.019837258,0.000054987107],"about_ca_topic_score_codex":0.0057365745,"about_ca_topic_score_gemma":0.009450947,"teacher_disagreement_score":0.005958419,"about_ca_system_score_codex":0.0010663505,"about_ca_system_score_gemma":0.0011069677,"threshold_uncertainty_score":0.021372616},"labels":[],"label_agreement":null},{"id":"W3168361240","doi":"10.1007/s10664-021-10111-4","title":"PRINS: scalable model inference for component-based system logs","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg","keywords":"Computer science; Scalability; Component (thermodynamics); Inference; Data mining; Software; Component-based software engineering; Software system; Process (computing); Implementation; Software engineering; Machine learning; Artificial intelligence; Database; Programming language","score_opus":0.024218276058589044,"score_gpt":0.26245946344478,"score_spread":0.23824118738619096,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3168361240","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03933208,0.0011438901,0.7631118,0.0006981502,0.00016794934,0.00043942948,0.009859082,0.18371575,0.0015318797],"genre_scores_gemma":[0.37393522,0.0005509359,0.59195966,0.00042764808,0.00014368788,0.000554942,0.027302114,0.003152454,0.0019733256],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99683243,0.0010070817,0.00026650194,0.00073200936,0.0009950751,0.00016684312],"domain_scores_gemma":[0.98465216,0.009857536,0.00088005135,0.0033818232,0.0009232231,0.00030531106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043404014,0.00266513,0.0013394362,0.003534628,0.00062632107,0.0019841252,0.004544325,0.0013739504,0.0039217616],"category_scores_gemma":[0.027470272,0.0013809208,0.0024224068,0.0023873174,0.000771341,0.0046651913,0.0028254637,0.0031865824,0.0017330468],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010730547,0.0009195539,0.037493717,0.0019155355,0.0017780905,0.00072134193,0.0005775682,0.5086245,0.010282122,0.015310882,0.06254857,0.3587551],"study_design_scores_gemma":[0.000033720884,0.000019497993,0.00063789985,0.0000128164575,0.000023283514,0.00004261617,0.000028060645,0.9880647,0.001231292,0.008335086,0.0015590364,0.000011920782],"about_ca_topic_score_codex":0.014362234,"about_ca_topic_score_gemma":0.027850354,"teacher_disagreement_score":0.014362234,"about_ca_system_score_codex":0.0012210197,"about_ca_system_score_gemma":0.0034038126,"threshold_uncertainty_score":0.0285573},"labels":[],"label_agreement":null},{"id":"W3176971075","doi":"10.1007/s10664-021-09994-0","title":"MLASP: Machine learning assisted capacity planning","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Top Hat (Canada)","funders":"","keywords":"Computer science; Capacity planning; Artificial intelligence; Operating system","score_opus":0.03252904706006558,"score_gpt":0.26101616788215914,"score_spread":0.22848712082209355,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176971075","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012478846,0.0003876792,0.8329028,0.0007517913,0.00038640428,0.00031207126,0.008658398,0.12811516,0.016006943],"genre_scores_gemma":[0.3348296,0.00025784184,0.6388623,0.00032713654,0.00021120477,0.0006476879,0.009240511,0.0038357056,0.011787991],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99941874,0.0001863264,0.000031269916,0.00010986488,0.00018261155,0.00007126655],"domain_scores_gemma":[0.99785703,0.0011164857,0.00012254245,0.00034770515,0.00041941446,0.0001367965],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010900326,0.0010886156,0.0007585208,0.0015939919,0.00048409702,0.0014805511,0.0020943934,0.0011220323,0.028489476],"category_scores_gemma":[0.006702336,0.00068356853,0.0008238209,0.0011264231,0.00041394203,0.0019371695,0.0014662776,0.0019277658,0.0071261907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036469018,0.0002720625,0.002006347,0.00029555545,0.0001274741,0.0002023197,0.00008880211,0.50540686,0.0016550857,0.017971143,0.13309652,0.3385131],"study_design_scores_gemma":[0.000030787123,0.000023035062,0.00018028161,0.000011170431,0.000008097789,0.000021765461,0.000010453935,0.9795782,0.001321149,0.013222841,0.0055804667,0.000011788159],"about_ca_topic_score_codex":0.007995283,"about_ca_topic_score_gemma":0.011723688,"teacher_disagreement_score":0.028489476,"about_ca_system_score_codex":0.0007497993,"about_ca_system_score_gemma":0.0019186977,"threshold_uncertainty_score":0.095306754},"labels":[],"label_agreement":null},{"id":"W3176989815","doi":"10.1007/s10664-021-10066-6","title":"Test case selection and prioritization using machine learning: a systematic literature review","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":158,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Mitacs; Huawei Technologies; Canada Research Chairs","keywords":"Regression testing; Computer science; Machine learning; Prioritization; Artificial intelligence; Feature selection; Process (computing); Software; Test case; Test (biology); Software engineering; Regression analysis; Software development; Engineering; Management science; Software construction","score_opus":0.019469488494802994,"score_gpt":0.28500089700866355,"score_spread":0.26553140851386053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176989815","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010676198,0.9761461,0.0068027824,0.0016495943,0.00025014917,0.0027337333,0.0006866548,0.00006691495,0.0009878216],"genre_scores_gemma":[0.12953013,0.8321085,0.030721243,0.0020184899,0.00027611342,0.0038158,0.0011527241,0.000069865826,0.00030709436],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.9552804,0.018287372,0.014796616,0.0027365715,0.008360227,0.00053867436],"domain_scores_gemma":[0.73209107,0.22594377,0.020362787,0.0047323997,0.015695766,0.0011741499],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05975972,0.0019583036,0.007467157,0.021022012,0.0012616508,0.003955563,0.004611086,0.002447627,0.003002858],"category_scores_gemma":[0.20527458,0.0014213604,0.006923026,0.01496554,0.00170537,0.0054163244,0.0026922813,0.0019000317,0.00039086648],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060780614,0.00033599694,0.006460019,0.53913754,0.007851888,0.0002643355,0.0011868357,0.0010629719,0.00056655996,0.000979983,0.0035188561,0.43802717],"study_design_scores_gemma":[0.00091627456,0.0009945198,0.010880734,0.8951286,0.050811745,0.001148266,0.0023542584,0.0024966463,0.0019649847,0.003350326,0.029751172,0.00020252314],"about_ca_topic_score_codex":0.0056227976,"about_ca_topic_score_gemma":0.015517925,"teacher_disagreement_score":0.94024026,"about_ca_system_score_codex":0.00584488,"about_ca_system_score_gemma":0.027835494,"threshold_uncertainty_score":0.31604338},"labels":[],"label_agreement":null},{"id":"W3178956269","doi":"10.1007/s10664-021-10004-6","title":"Evaluating the impact of falsely detected performance bug-inducing changes in JIT models","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Polytechnique Montréal; Concordia University","funders":"","keywords":"Leverage (statistics); Software bug; Computer science; Software; Software quality; Empirical research; Software quality assurance; Quality (philosophy); Capability Maturity Model; Source code; Software engineering; Software development; Operating system; Artificial intelligence","score_opus":0.09454756759896381,"score_gpt":0.3660511802990926,"score_spread":0.2715036127001288,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3178956269","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99078673,0.00033163384,0.0067620757,0.00029367878,0.000079418714,0.000032979995,0.000305711,0.0004710425,0.00093670975],"genre_scores_gemma":[0.99392796,0.000047624133,0.0052637956,0.000053583874,0.000019087434,0.0000118137295,0.00038800802,0.00006602788,0.0002221672],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98631895,0.0069007315,0.0009060206,0.0022665448,0.002918123,0.0006895311],"domain_scores_gemma":[0.5707826,0.383166,0.016538428,0.018204713,0.009001589,0.0023067065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015667267,0.0010076228,0.0006448692,0.0014643052,0.00063574995,0.0015359956,0.001591823,0.0020867425,0.0014371319],"category_scores_gemma":[0.1943422,0.0005431952,0.0010639754,0.0008711187,0.0013059058,0.0023500277,0.001165968,0.0021726678,0.000258705],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006643844,0.0027005475,0.47332674,0.0007251075,0.001526198,0.0006788447,0.0007320732,0.37685367,0.0149976425,0.0038857672,0.003050113,0.114879444],"study_design_scores_gemma":[0.00026046872,0.0031243344,0.10073688,0.00010146585,0.0008157821,0.00040667277,0.00034821924,0.8775497,0.0120962,0.0036918267,0.00078738574,0.000081086255],"about_ca_topic_score_codex":0.0069970135,"about_ca_topic_score_gemma":0.011818459,"teacher_disagreement_score":0.015667267,"about_ca_system_score_codex":0.0012678988,"about_ca_system_score_gemma":0.0018331595,"threshold_uncertainty_score":0.08285743},"labels":[],"label_agreement":null},{"id":"W3179948295","doi":"10.1007/s10664-021-10006-4","title":"Task estimation for software company employees based on computer interaction logs","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Business Process Modeling and Analysis","field":"Business, Management and Accounting","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Task (project management); Computer science; Software; Estimation; Software engineering; Engineering; Operating system; Systems engineering","score_opus":0.02722979127888976,"score_gpt":0.2638586404644538,"score_spread":0.23662884918556404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3179948295","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9829754,0.000062524094,0.015505182,0.000036582038,0.000007030687,0.000038122176,0.0004608123,0.00033937313,0.0005749844],"genre_scores_gemma":[0.9955188,0.000030660252,0.0033013015,0.000007199253,0.0000064862124,0.000031764957,0.0006955287,0.000018374334,0.0003897373],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9991447,0.00032245024,0.00008055954,0.00012307138,0.00022617563,0.0001030728],"domain_scores_gemma":[0.9787338,0.015782898,0.001791088,0.0010296562,0.002126848,0.0005357834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013813726,0.00042094442,0.00037558106,0.0016219922,0.00022766486,0.00077208376,0.00040444275,0.00068837346,0.0012358492],"category_scores_gemma":[0.017457118,0.00022707995,0.00032782299,0.00083370565,0.00012361904,0.0008172737,0.00040124994,0.0005105048,0.00067737274],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025671707,0.0014795153,0.73160356,0.00021671319,0.00017454478,0.00018192534,0.0012403174,0.039889082,0.012401554,0.0004907223,0.0020357901,0.20771906],"study_design_scores_gemma":[0.00005701742,0.00084205077,0.4445291,0.000032458036,0.000096787175,0.0001353933,0.00068298535,0.5451559,0.006830216,0.00092171866,0.0006521861,0.00006433952],"about_ca_topic_score_codex":0.0073965243,"about_ca_topic_score_gemma":0.0068041957,"teacher_disagreement_score":0.0073965243,"about_ca_system_score_codex":0.00038720918,"about_ca_system_score_gemma":0.0005325066,"threshold_uncertainty_score":0.01470691},"labels":[],"label_agreement":null},{"id":"W3180512116","doi":"10.1007/s10664-021-09969-1","title":"The secret life of test smells - an empirical study on test smell evolution and maintenance","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Code smell; Test (biology); Code refactoring; Empirical research; Computer science; Reliability engineering; Engineering; Software; Software quality; Software development; Statistics","score_opus":0.02457635823879043,"score_gpt":0.29435137768838804,"score_spread":0.2697750194495976,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3180512116","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.998623,0.00015588295,0.00039297898,0.00007426805,0.0000026004252,0.0000052373966,0.000060296756,0.000011228198,0.00067444495],"genre_scores_gemma":[0.99955696,0.000031143227,0.00012783123,0.000011847089,0.0000037980435,0.0000032200578,0.00008725509,0.000006584931,0.00017136421],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9974942,0.00079902285,0.00025905643,0.00021905715,0.0010425512,0.0001860485],"domain_scores_gemma":[0.8514193,0.08641453,0.04217948,0.0071879686,0.006651992,0.006146668],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0047808522,0.00018416664,0.00024103232,0.0023676832,0.00071941095,0.0015299442,0.0007630025,0.00082968245,0.0016562941],"category_scores_gemma":[0.065046534,0.00021438388,0.0004089064,0.0018878623,0.0014472421,0.004139775,0.001699617,0.0013061739,0.00032737668],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030094106,0.00045261512,0.96782255,0.00006115201,0.00006195097,0.00035286398,0.005216712,0.00055330206,0.001419418,0.0014746253,0.0003333651,0.021950584],"study_design_scores_gemma":[0.000013418273,0.0004279023,0.98616064,0.000053194693,0.00003535335,0.000662176,0.0043838364,0.0040376415,0.0010552513,0.0021536273,0.0009839991,0.000033053253],"about_ca_topic_score_codex":0.0018281317,"about_ca_topic_score_gemma":0.002583406,"teacher_disagreement_score":0.9952192,"about_ca_system_score_codex":0.00082396145,"about_ca_system_score_gemma":0.00056980253,"threshold_uncertainty_score":0.025283873},"labels":[],"label_agreement":null},{"id":"W3185088259","doi":"10.1007/s10664-021-09992-2","title":"Perceived diversity in software engineering: a systematic literature review","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":136,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of British Columbia, Okanagan Campus; Kelowna General Hospital; University of British Columbia","funders":"","keywords":"Diversity (politics); Gender diversity; Diversity training; Cultural diversity; Knowledge management; Computer science; Psychology; Engineering; Political science; Management","score_opus":0.023105166104566845,"score_gpt":0.2595688079767733,"score_spread":0.23646364187220645,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3185088259","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018443739,0.97726214,0.00092960946,0.0010550278,0.00014819409,0.0005873853,0.0006248309,0.00001051693,0.00093865336],"genre_scores_gemma":[0.118886836,0.8747106,0.0031854538,0.0015774057,0.00013905158,0.00082148367,0.0005085592,0.000014043484,0.00015655684],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.97926587,0.0070786835,0.0070505417,0.0017025301,0.0044496264,0.00045289466],"domain_scores_gemma":[0.8307122,0.14076582,0.01481803,0.0017666584,0.010533592,0.0014037552],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.026049469,0.00083631586,0.004214243,0.018953035,0.0010170877,0.0041696425,0.0016088369,0.0020885395,0.0028272767],"category_scores_gemma":[0.11045239,0.0009656984,0.004342409,0.014789347,0.0017572516,0.0051004305,0.002867959,0.0017052605,0.00024932236],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032359874,0.00021396017,0.025280228,0.74503285,0.00905435,0.000337829,0.005007915,0.00028096535,0.00038249986,0.0012887609,0.0025553755,0.21024172],"study_design_scores_gemma":[0.00025898303,0.00038457057,0.044177014,0.89010155,0.032425378,0.0008210566,0.0075820517,0.00023018176,0.00029908225,0.0013705454,0.022222947,0.00012667853],"about_ca_topic_score_codex":0.0070493915,"about_ca_topic_score_gemma":0.026401674,"teacher_disagreement_score":0.9739505,"about_ca_system_score_codex":0.004114368,"about_ca_system_score_gemma":0.020391257,"threshold_uncertainty_score":0.1377644},"labels":[],"label_agreement":null},{"id":"W3185295980","doi":"10.1007/s10664-021-10099-x","title":"Clones in deep learning code: what, where, and why?","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Deep learning; Computer science; Artificial intelligence; Timeline; Source lines of code; Context (archaeology); Python (programming language); Dependability; Software development; Java; Software evolution; Software system; Software engineering; Machine learning; Software quality; Code (set theory); Software; Programming language; Software construction; Biology","score_opus":0.017447204121601945,"score_gpt":0.26496860410912637,"score_spread":0.24752139998752443,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3185295980","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9512247,0.0013385043,0.039841104,0.0036812627,0.000065420296,0.000028881075,0.0001635042,0.00034169524,0.003315028],"genre_scores_gemma":[0.9914062,0.00022959086,0.006951903,0.00022584996,0.000030777475,0.000017398514,0.000107738306,0.00011067392,0.0009199992],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9934296,0.002333936,0.0003049168,0.0011222424,0.0022838723,0.0005254023],"domain_scores_gemma":[0.8946272,0.072831996,0.010017982,0.012021314,0.008804229,0.0016973104],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062619396,0.0003959183,0.0005534513,0.0013968645,0.001075719,0.002350294,0.0011174148,0.0022338745,0.0024517518],"category_scores_gemma":[0.12980126,0.0005521412,0.0004591212,0.001920214,0.0055271336,0.010591605,0.0023267716,0.0031851744,0.00038329104],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042183712,0.000358155,0.5429724,0.00034046732,0.00019705975,0.0005441908,0.0075585553,0.022031022,0.0045865704,0.09009612,0.0066610095,0.3242327],"study_design_scores_gemma":[0.0001375919,0.00055900315,0.19112568,0.00088022626,0.0003808433,0.0020739737,0.0090963645,0.27042732,0.019516613,0.49325302,0.012379593,0.00016973504],"about_ca_topic_score_codex":0.0063897497,"about_ca_topic_score_gemma":0.009067275,"teacher_disagreement_score":0.0063897497,"about_ca_system_score_codex":0.002328058,"about_ca_system_score_gemma":0.0021302428,"threshold_uncertainty_score":0.0331167},"labels":[],"label_agreement":null},{"id":"W3192457683","doi":"10.1007/s10664-021-09980-6","title":"An empirical study of same-day releases of popular packages in the npm ecosystem","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Alberta; Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Reuse; Schedule; Exploratory research; Software engineering; Operating system; Engineering","score_opus":0.030237698177021738,"score_gpt":0.3166212142887117,"score_spread":0.28638351611169,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3192457683","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99908435,0.000021968011,0.00006909941,0.00005060438,0.0000016691937,0.0000062125036,0.00007301123,0.000005373885,0.0006876795],"genre_scores_gemma":[0.99861956,0.000033167416,0.00024915143,0.000035938432,0.000007264933,0.000012953367,0.00030244948,0.000009188328,0.00073039863],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9979109,0.0007876703,0.00011874274,0.000320137,0.00057508826,0.00028750638],"domain_scores_gemma":[0.94000894,0.032847553,0.013727369,0.0033966608,0.0056047128,0.0044147316],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031788987,0.00023750436,0.00030485707,0.0020154219,0.0013684161,0.0019552566,0.0009649773,0.0011830695,0.0026129917],"category_scores_gemma":[0.033366907,0.0003274307,0.00030046506,0.002458965,0.0014762221,0.0039806003,0.0016173882,0.0019518937,0.000719413],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043151423,0.0020423615,0.97087437,0.00006253525,0.000059663646,0.000529864,0.008188143,0.00034215307,0.0014927596,0.0011495486,0.001125786,0.013701227],"study_design_scores_gemma":[0.000016431726,0.00026226754,0.98731893,0.000017701448,0.000018040906,0.000251652,0.008226448,0.0022071237,0.00027955,0.00024108765,0.0011405336,0.000020327157],"about_ca_topic_score_codex":0.011871191,"about_ca_topic_score_gemma":0.024066865,"teacher_disagreement_score":0.011871191,"about_ca_system_score_codex":0.0013823682,"about_ca_system_score_gemma":0.00081731926,"threshold_uncertainty_score":0.023604155},"labels":[],"label_agreement":null},{"id":"W3193822405","doi":"10.1007/s10664-021-10014-4","title":"An empirical study of Q&amp;A websites for game developers","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Expert finding and Q&A systems","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Latent Dirichlet allocation; Game Developer; Game testing; Computer science; Video game development; Empirical research; World Wide Web; Game design document; Software development; Software; Game design; Topic model; Data science; Knowledge management; Multimedia; Artificial intelligence","score_opus":0.05243780047764452,"score_gpt":0.3366985135992673,"score_spread":0.28426071312162277,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3193822405","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9964598,0.000053800657,0.00013817093,0.00020602727,0.0000030427286,0.00006476343,0.000054566914,0.000009234849,0.0030104956],"genre_scores_gemma":[0.9984199,0.000059517923,0.00028052906,0.00009560326,0.00000439564,0.00004936092,0.00007328639,0.000009881715,0.0010074982],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99448895,0.0030845015,0.0003264051,0.00047747887,0.001029184,0.00059350714],"domain_scores_gemma":[0.68960786,0.25147817,0.023229875,0.0063988324,0.016633,0.012652271],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008737872,0.00029928033,0.00035989477,0.0036390128,0.003673685,0.0038847958,0.0012749217,0.0018109856,0.005646696],"category_scores_gemma":[0.10554287,0.0005690907,0.00020607498,0.0025012076,0.0023641381,0.00482927,0.002486493,0.0028854378,0.0011008658],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008353746,0.015004643,0.840322,0.0003687468,0.00004687084,0.0014971837,0.09555512,0.00023035744,0.0011567724,0.0039214697,0.0033552551,0.037706085],"study_design_scores_gemma":[0.00023004977,0.0019216648,0.80208445,0.00028563014,0.00007214027,0.0009344169,0.1793551,0.003450623,0.0014485726,0.0011906183,0.008939009,0.00008773004],"about_ca_topic_score_codex":0.0129364515,"about_ca_topic_score_gemma":0.02285461,"teacher_disagreement_score":0.0129364515,"about_ca_system_score_codex":0.0026139389,"about_ca_system_score_gemma":0.0044300514,"threshold_uncertainty_score":0.046210885},"labels":[],"label_agreement":null},{"id":"W3194114348","doi":"10.1007/s10664-021-09986-0","title":"Empirical evaluation of tools for hairy requirements engineering tasks","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Empirical research; Systems engineering; Software engineering; Engineering; Mathematics","score_opus":0.13428343853465402,"score_gpt":0.37559869310442306,"score_spread":0.24131525456976904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3194114348","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9883222,0.00016914672,0.00805306,0.000110920344,0.000022855633,0.00037983354,0.00015714044,0.00047951762,0.002305364],"genre_scores_gemma":[0.97550327,0.00015146339,0.021923466,0.00007815606,0.000014789796,0.0004066977,0.00057671434,0.00014684997,0.0011985728],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9840821,0.009330714,0.0013851408,0.0009633723,0.0036604754,0.00057821436],"domain_scores_gemma":[0.6508252,0.30127162,0.011887866,0.016908186,0.0154944975,0.003612624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016138071,0.0008280085,0.00048132226,0.002902193,0.0007698184,0.001945292,0.0019958273,0.0013307058,0.0030329246],"category_scores_gemma":[0.17741063,0.0004990667,0.00052271114,0.0015848774,0.0010305081,0.0027361743,0.0029392568,0.001194169,0.0009403788],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013083866,0.022270802,0.09134758,0.0043714345,0.000404185,0.0010657054,0.025707634,0.020671995,0.035250485,0.0047894316,0.0065940944,0.7744428],"study_design_scores_gemma":[0.0084390445,0.06642298,0.5210679,0.002815084,0.0011117549,0.0019191714,0.028499957,0.25890565,0.06279161,0.00939551,0.038017448,0.0006139723],"about_ca_topic_score_codex":0.0019227756,"about_ca_topic_score_gemma":0.0031549537,"teacher_disagreement_score":0.016138071,"about_ca_system_score_codex":0.0012926715,"about_ca_system_score_gemma":0.0015704982,"threshold_uncertainty_score":0.085347354},"labels":[],"label_agreement":null},{"id":"W3195715707","doi":"10.1007/s10664-022-10136-3","title":"The sense of logging in the Linux kernel","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Conselho Nacional de Desenvolvimento Científico e Tecnológico; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Mitacs; Coordenação de Aperfeiçoamento de Pessoal de Nível Superior","keywords":"Logging; Computer science; Linux kernel; Debugging; Operating system; Source code; Java; Consistency (knowledge bases); Database; Forestry","score_opus":0.013708953952654526,"score_gpt":0.24621354922894564,"score_spread":0.2325045952762911,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3195715707","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89496756,0.0022734355,0.032398697,0.014867702,0.0002469972,0.00002299896,0.00017307932,0.00005509308,0.054994516],"genre_scores_gemma":[0.9987544,0.00014520081,0.00060769665,0.00020166885,0.000046890775,0.0000044535,0.000013439828,0.0000091360625,0.0002170916],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9956839,0.00281646,0.00021907309,0.00044117114,0.0005912418,0.00024810317],"domain_scores_gemma":[0.94642675,0.038455997,0.0062907827,0.004936171,0.0020817495,0.0018086208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060512205,0.00021550391,0.0002660623,0.0018441606,0.001468061,0.00597222,0.0007712943,0.0014873006,0.0024714402],"category_scores_gemma":[0.06270962,0.00036765402,0.00020121856,0.0019578647,0.016156048,0.011409661,0.0031513104,0.0028130435,0.00015470613],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006105375,0.00022692281,0.070642695,0.0002518059,0.00006913255,0.00031991265,0.0540316,0.002392132,0.0019485229,0.82509565,0.0025799633,0.041831136],"study_design_scores_gemma":[0.000047813606,0.00014157506,0.066248454,0.00021974489,0.000045348606,0.00049046206,0.030793445,0.0050494466,0.0005133471,0.88913006,0.0072374875,0.00008280413],"about_ca_topic_score_codex":0.0015242624,"about_ca_topic_score_gemma":0.0012527729,"teacher_disagreement_score":0.0060512205,"about_ca_system_score_codex":0.0009279644,"about_ca_system_score_gemma":0.0009188591,"threshold_uncertainty_score":0.03200233},"labels":[],"label_agreement":null},{"id":"W3198375663","doi":"10.1007/s10664-021-10021-5","title":"An empirical study of IoT topics in IoT developer discussions on Stack Overflow","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Trent University; Concordia University; University of Calgary","funders":"","keywords":"Computer science; Internet of Things; Popularity; World Wide Web; Categorization; Software; Download; Data science; Multimedia; Artificial intelligence","score_opus":0.025234064105648206,"score_gpt":0.2996113619742789,"score_spread":0.2743772978686307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3198375663","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.992939,0.00006193655,0.0006127443,0.00042227778,0.000015048487,0.00006089937,0.00013948648,0.000018427103,0.0057302],"genre_scores_gemma":[0.99741024,0.00006954894,0.0004509848,0.00013443307,0.000027886636,0.000093073606,0.00014570876,0.000018580224,0.001649541],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9937378,0.003431237,0.00034610587,0.0006275209,0.0011649918,0.0006923643],"domain_scores_gemma":[0.8304044,0.13349555,0.018928012,0.0028243249,0.008976298,0.0053714523],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.009958608,0.0003972131,0.0002788454,0.004979703,0.0035719464,0.0033355292,0.0008411935,0.0014970867,0.003908815],"category_scores_gemma":[0.076779775,0.00042218843,0.00026962615,0.0038707696,0.002151846,0.005031886,0.0034863737,0.0020250336,0.0005908486],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050257833,0.0013348719,0.60149014,0.00026176518,0.000059416958,0.0008405614,0.35361648,0.0004438634,0.0025148506,0.005657025,0.003342303,0.029936071],"study_design_scores_gemma":[0.00007365204,0.00040628132,0.48418903,0.00030224226,0.00006412538,0.00021796628,0.48754072,0.0045103296,0.002058807,0.0030905306,0.01744813,0.0000980925],"about_ca_topic_score_codex":0.018938592,"about_ca_topic_score_gemma":0.024772871,"teacher_disagreement_score":0.9950203,"about_ca_system_score_codex":0.003217793,"about_ca_system_score_gemma":0.0031996425,"threshold_uncertainty_score":0.052666783},"labels":[],"label_agreement":null},{"id":"W3200568639","doi":"10.1007/s10664-021-10025-1","title":"A study of how Docker Compose is used to compose multi-component systems","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Component (thermodynamics); Computer science; Container (type theory); Software deployment; Operating system; Component-based software engineering; Software; Leverage (statistics); Web application; Software engineering; AKA; Database; World Wide Web; Software system; Engineering","score_opus":0.07160242864086612,"score_gpt":0.3200402561237651,"score_spread":0.24843782748289894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3200568639","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9844694,0.000048532034,0.008762855,0.00010753305,0.000009030151,0.000039150444,0.000020827889,0.00015556399,0.0063871257],"genre_scores_gemma":[0.9843068,0.000056217435,0.011941672,0.000036836595,0.0000049151536,0.000016487294,0.000072413255,0.00010822346,0.0034564196],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9980824,0.0009085393,0.0000970707,0.00021114896,0.0005715563,0.00012924412],"domain_scores_gemma":[0.9688757,0.024654223,0.001448322,0.002756404,0.0017696572,0.0004956226],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032428196,0.00038636624,0.00033318633,0.0012402341,0.0015881654,0.002044312,0.00083639077,0.0009310951,0.0035446247],"category_scores_gemma":[0.03268315,0.00039132318,0.0003573215,0.0014980414,0.0013103655,0.0034623654,0.0010345707,0.0011280978,0.00047845364],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017378045,0.0065450165,0.31472546,0.00075664197,0.00028603434,0.0031858492,0.110269345,0.03303064,0.0545902,0.053149592,0.0060680853,0.4156554],"study_design_scores_gemma":[0.00031662363,0.004083628,0.38680056,0.00028710792,0.00043855116,0.0041059162,0.10189723,0.3354302,0.080495186,0.025371552,0.060454525,0.00031892088],"about_ca_topic_score_codex":0.009062455,"about_ca_topic_score_gemma":0.015047413,"teacher_disagreement_score":0.009062455,"about_ca_system_score_codex":0.0014104847,"about_ca_system_score_gemma":0.0010326564,"threshold_uncertainty_score":0.018019438},"labels":[],"label_agreement":null},{"id":"W3202264790","doi":"10.1007/s10664-021-10016-2","title":"Rotten green tests in Java, Pharo and Python","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Python (programming language); Java; Computer science; Programming language; Unit testing; Test (biology); Software; Biology; Ecology","score_opus":0.02234213422314712,"score_gpt":0.284636052570599,"score_spread":0.26229391834745186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3202264790","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90499026,0.00042547032,0.05048542,0.0007602078,0.00023979954,0.00013034964,0.0027823427,0.001349985,0.038836107],"genre_scores_gemma":[0.9804277,0.000069485795,0.011090555,0.00017001558,0.000056616467,0.00020652187,0.0018337141,0.00079454057,0.0053508803],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98426497,0.009902264,0.0010352904,0.0017412667,0.0020362644,0.0010200508],"domain_scores_gemma":[0.6003603,0.3712649,0.0082448665,0.011724281,0.005660813,0.0027449636],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01407781,0.0007869022,0.0009235585,0.0021881901,0.0011575933,0.0021441765,0.0021470136,0.001102272,0.01667463],"category_scores_gemma":[0.16344479,0.00037010547,0.0017472012,0.0034462004,0.0022645323,0.006842692,0.0030529378,0.002629067,0.0022032284],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.012389815,0.0042234124,0.3444031,0.0014348704,0.0021696433,0.00078247744,0.008722796,0.06147906,0.0037093365,0.19967547,0.05490549,0.3061045],"study_design_scores_gemma":[0.0011823855,0.005191617,0.5143338,0.0004995439,0.0010991392,0.00046054213,0.008830825,0.26784265,0.0098702945,0.16028425,0.030072125,0.00033283385],"about_ca_topic_score_codex":0.014387677,"about_ca_topic_score_gemma":0.014646295,"teacher_disagreement_score":0.9859222,"about_ca_system_score_codex":0.0013408873,"about_ca_system_score_gemma":0.0020776817,"threshold_uncertainty_score":0.074451506},"labels":[],"label_agreement":null},{"id":"W3204686189","doi":"10.1007/s10664-021-10032-2","title":"How are project-specific forums utilized? A study of participation, content, and sentiment in the Eclipse ecosystem","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Japan Society for the Promotion of Science","keywords":"Eclipse; Popularity; World Wide Web; Computer science; Bridge (graph theory); Sentiment analysis; Knowledge management; Data science; Public relations; Political science; Artificial intelligence","score_opus":0.09961188966302743,"score_gpt":0.31488797726991474,"score_spread":0.21527608760688732,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3204686189","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.998145,0.00002943616,0.0004636102,0.00007167508,0.000004523235,0.000010875023,0.000025024732,0.000007161635,0.0012427428],"genre_scores_gemma":[0.9992894,0.000023101325,0.00029837477,0.00001780747,0.000006234905,0.000013888417,0.000046770216,0.000005973174,0.00029853728],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9960687,0.0024388663,0.00019818523,0.00037231387,0.00052233326,0.00039956666],"domain_scores_gemma":[0.96954465,0.016048491,0.0077972203,0.0011179665,0.0031130386,0.002378726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006973767,0.00020108696,0.0002571812,0.0021035331,0.0015154429,0.0023807054,0.00031111878,0.00055035006,0.0012981656],"category_scores_gemma":[0.022971151,0.00019822957,0.0002218682,0.0013163672,0.0008317422,0.0027798698,0.0019463735,0.00052935537,0.00020839354],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023500566,0.0001862105,0.85319203,0.00010635799,0.000042628173,0.00040202242,0.10220126,0.0001268336,0.0049561444,0.00089200865,0.0007209975,0.036938358],"study_design_scores_gemma":[0.00001316409,0.0002093494,0.88727355,0.0000853582,0.000033958455,0.00027600225,0.10184477,0.002074606,0.0009705388,0.000684242,0.006500919,0.00003354081],"about_ca_topic_score_codex":0.0016624271,"about_ca_topic_score_gemma":0.0026116674,"teacher_disagreement_score":0.006973767,"about_ca_system_score_codex":0.00076564756,"about_ca_system_score_gemma":0.0005088819,"threshold_uncertainty_score":0.036881268},"labels":[],"label_agreement":null},{"id":"W3206397610","doi":"10.1007/s10664-021-10083-5","title":"A fine-grained data set and analysis of tangling in bug fixing commits","year":2022,"lang":"en","type":"preprint","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Trent University; University of Saskatchewan; University of British Columbia, Okanagan Campus; Kelowna General Hospital; University of British Columbia; University of Ottawa","funders":"Horizon 2020 Framework Programme; Technische Universität Clausthal; Deutsche Forschungsgemeinschaft","keywords":"Context (archaeology); Computer science; Software bug; Set (abstract data type); Code (set theory); Software; Source lines of code; Data mining; Programming language; Biology","score_opus":0.07673165410839632,"score_gpt":0.34358802997785276,"score_spread":0.26685637586945643,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3206397610","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97008944,0.00052561867,0.011073269,0.00036167077,0.000048331236,0.0002364,0.015461462,0.00037038172,0.0018334168],"genre_scores_gemma":[0.96159846,0.00009102571,0.013655967,0.00009147257,0.000037888716,0.00037433644,0.023522386,0.00011425798,0.0005141945],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.979675,0.0075764,0.002795614,0.0040859007,0.005221998,0.0006451155],"domain_scores_gemma":[0.68688923,0.19615611,0.045545705,0.03565111,0.03193932,0.0038185245],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018041318,0.00046662518,0.00065450167,0.01178225,0.0011125803,0.001573871,0.0009788362,0.0014889835,0.0014291458],"category_scores_gemma":[0.13322534,0.00038103666,0.0004849198,0.008459627,0.001358744,0.0019275895,0.0024554983,0.0012980315,0.0007967242],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048055997,0.00041443214,0.9098739,0.000916429,0.00025864824,0.0007150537,0.0071425973,0.0064129042,0.0039795563,0.0017985372,0.011520951,0.05648641],"study_design_scores_gemma":[0.000039527506,0.00021282928,0.96428853,0.00027806446,0.000067518,0.00046296392,0.0023494896,0.015844401,0.0025029955,0.002293041,0.011576168,0.000084614534],"about_ca_topic_score_codex":0.0064612944,"about_ca_topic_score_gemma":0.007807063,"teacher_disagreement_score":0.018041318,"about_ca_system_score_codex":0.0010879164,"about_ca_system_score_gemma":0.0010803821,"threshold_uncertainty_score":0.09541273},"labels":[],"label_agreement":null},{"id":"W3208323417","doi":"10.1007/s10664-021-10045-x","title":"How do i refactor this? An empirical study on refactoring trends and topics in Stack Overflow","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":64,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"","keywords":"Code refactoring; Computer science; Software engineering; Unit testing; Empirical research; Implementation; Software; Programming language","score_opus":0.04809255609342547,"score_gpt":0.3290085982567614,"score_spread":0.2809160421633359,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3208323417","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9983246,0.00022549568,0.00045690036,0.00021135576,0.0000069520843,0.00002064588,0.00009046943,0.000030928888,0.00063272624],"genre_scores_gemma":[0.99664706,0.00043674835,0.0017621474,0.00011728238,0.000021870512,0.000025828072,0.00036267424,0.000036981517,0.0005893994],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99494874,0.0013970932,0.00070186035,0.00060402445,0.0019773003,0.0003709095],"domain_scores_gemma":[0.7850826,0.12681244,0.05749231,0.006175078,0.01851883,0.005918774],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.007555335,0.00033830566,0.0003359011,0.0041468325,0.0010443691,0.0021359874,0.0010238556,0.0011564542,0.0008664238],"category_scores_gemma":[0.121244974,0.00047086208,0.00043438136,0.003695196,0.0009257509,0.004660071,0.001093143,0.0021360088,0.0003090821],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020484076,0.00056143495,0.91184765,0.00018524042,0.00007731711,0.0003566733,0.028118916,0.000255124,0.0027551565,0.0004112467,0.00072659756,0.054499906],"study_design_scores_gemma":[0.000019049357,0.00050432136,0.9635402,0.00021513175,0.00012436356,0.00065992924,0.027272455,0.0021178343,0.002214794,0.0005993918,0.0026657234,0.000066694105],"about_ca_topic_score_codex":0.0062571196,"about_ca_topic_score_gemma":0.008850445,"teacher_disagreement_score":0.9924447,"about_ca_system_score_codex":0.0013392511,"about_ca_system_score_gemma":0.0024550776,"threshold_uncertainty_score":0.039956927},"labels":[],"label_agreement":null},{"id":"W4200448782","doi":"10.1007/s10664-021-10076-4","title":"Using code reviews to automatically configure static analysis tools","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Leverage (statistics); Code review; Static program analysis; Java; Code (set theory); Source code; Precision and recall; Context (archaeology); Information retrieval; Software engineering; Statement (logic); Programming language; Artificial intelligence; Software; Software development","score_opus":0.09646917508994118,"score_gpt":0.3629266213116876,"score_spread":0.26645744622174644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200448782","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43190795,0.001980081,0.38422498,0.0012347674,0.0005324467,0.0012065603,0.0035547805,0.1573122,0.01804623],"genre_scores_gemma":[0.69224,0.00037115795,0.28646386,0.0002745926,0.00016100264,0.0005091701,0.005263946,0.0074126185,0.0073036146],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9886114,0.0037619763,0.00089229766,0.0021777104,0.004205462,0.00035123108],"domain_scores_gemma":[0.8717375,0.06693242,0.016523622,0.013710327,0.028865745,0.002230371],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0063219476,0.001538462,0.0010644357,0.009929089,0.0009738647,0.002639444,0.001645864,0.0010762541,0.0038429638],"category_scores_gemma":[0.08582751,0.0010322994,0.0006241817,0.0028378936,0.0003782726,0.0025604372,0.0017988621,0.001008924,0.0036700617],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00056595076,0.00041241408,0.041433558,0.0007357438,0.00020370302,0.0005077586,0.0017176459,0.007318327,0.032192867,0.0018343468,0.038972877,0.8741047],"study_design_scores_gemma":[0.0006662958,0.0014759668,0.09610696,0.00098059,0.000689658,0.0019343899,0.002100208,0.67960757,0.1081542,0.011643512,0.096043296,0.0005973235],"about_ca_topic_score_codex":0.004487311,"about_ca_topic_score_gemma":0.0111260675,"teacher_disagreement_score":0.009929089,"about_ca_system_score_codex":0.0011304801,"about_ca_system_score_gemma":0.0031409257,"threshold_uncertainty_score":0.033434093},"labels":[],"label_agreement":null},{"id":"W4200462088","doi":"10.1007/s10664-021-10080-8","title":"A study of gender in user reviews on the Google Play Store","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Digital Marketing and Social Media","field":"Social Sciences","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"App store; Android (operating system); World Wide Web; Internet privacy; Mobile apps; Computer science; Social media; Advertising; Business","score_opus":0.0853712131898278,"score_gpt":0.3420812369426887,"score_spread":0.2567100237528609,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200462088","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9981692,0.00006532358,0.000029712177,0.00007382138,0.000005544116,0.0000072373623,0.000051633477,0.0000019036213,0.0015955305],"genre_scores_gemma":[0.9980349,0.000076903576,0.000047439768,0.00005244763,0.000010769496,0.000008217829,0.00008251619,0.000006168021,0.0016806459],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99866545,0.0005802134,0.000069988266,0.00012550724,0.00037616087,0.00018259036],"domain_scores_gemma":[0.973142,0.016189901,0.004240256,0.00047125047,0.004322997,0.0016334652],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016520818,0.00022452595,0.00029334152,0.001775826,0.0018606936,0.0020990367,0.00037223153,0.0006169069,0.0031411438],"category_scores_gemma":[0.017601619,0.00023759768,0.00023398943,0.0018034738,0.00078168896,0.0018555429,0.00083393295,0.0006723865,0.0005430627],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006242446,0.00067561073,0.8772065,0.000100053694,0.00005614195,0.0005848656,0.09647881,0.00005387515,0.0016433516,0.00057346246,0.0018979913,0.02010513],"study_design_scores_gemma":[0.000009456686,0.00024769927,0.9278853,0.000039975643,0.00003127525,0.00020273717,0.06766383,0.00036562604,0.00040459228,0.00007420844,0.0030479797,0.000027433567],"about_ca_topic_score_codex":0.04427071,"about_ca_topic_score_gemma":0.09930455,"teacher_disagreement_score":0.04427071,"about_ca_system_score_codex":0.0016085084,"about_ca_system_score_gemma":0.0009857579,"threshold_uncertainty_score":0.08802605},"labels":[],"label_agreement":null},{"id":"W4214741785","doi":"10.1007/s10664-021-10070-w","title":"TraceSim: An Alignment Method for Computing Stack Trace Similarity","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Telefonaktiebolaget LM Ericsson; Western Canada Research Grid; Compute Canada","keywords":"Computer science; Data deduplication; Crash; Data mining; TRACE (psycholinguistics); Software; Task (project management); Subroutine; Flexibility (engineering); Benchmark (surveying); Similarity (geometry); Information retrieval; Artificial intelligence; Machine learning; Database; Programming language; Engineering","score_opus":0.04467345933549554,"score_gpt":0.34524277005109066,"score_spread":0.3005693107155951,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4214741785","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014188775,0.00024596552,0.9434972,0.0000660146,0.00017599195,0.00021084782,0.002769149,0.03731395,0.0015321407],"genre_scores_gemma":[0.09748919,0.00024032935,0.88075626,0.00008230707,0.00008475869,0.0006209624,0.0111682825,0.0059861634,0.0035717788],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9964282,0.0005203065,0.00046416646,0.0008275076,0.0014961144,0.00026371545],"domain_scores_gemma":[0.99485165,0.001499495,0.0005686981,0.0012430171,0.0015981591,0.00023897541],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021726515,0.002268631,0.0016876182,0.009886033,0.0016234663,0.0030118262,0.003446664,0.0018681084,0.011372411],"category_scores_gemma":[0.018703416,0.0010830738,0.0016724197,0.010464676,0.00079746934,0.0052583176,0.0032683904,0.002578048,0.0058603836],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010529498,0.0005270039,0.01094595,0.0012482095,0.00072573277,0.00038105214,0.0012909959,0.023624498,0.034306686,0.022538656,0.04449526,0.85886306],"study_design_scores_gemma":[0.00030146693,0.0007383208,0.009259624,0.0002364883,0.00043878326,0.0011676252,0.0013693951,0.744424,0.08512905,0.0792137,0.07738351,0.0003380487],"about_ca_topic_score_codex":0.005397487,"about_ca_topic_score_gemma":0.009062468,"teacher_disagreement_score":0.011372411,"about_ca_system_score_codex":0.00083742384,"about_ca_system_score_gemma":0.002839408,"threshold_uncertainty_score":0.038044512},"labels":[],"label_agreement":null},{"id":"W4214871480","doi":"10.1007/s10664-021-10078-2","title":"Reuse and maintenance practices among divergent forks in three software ecosystems","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Polytechnique Montréal","funders":"Vetenskapsrådet; Vlaamse regering; Fonds De La Recherche Scientifique - FNRS; Fonds Wetenschappelijk Onderzoek; Canada Research Chairs","keywords":"Computer science; Software engineering; Software development; Software construction; Software bug; Software; Social software engineering; Software evolution; Software analytics; Software peer review; Software system; Code reuse; Software maintenance; Operating system","score_opus":0.02947488547754301,"score_gpt":0.27531612184347026,"score_spread":0.24584123636592725,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4214871480","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9951308,0.00014687103,0.0038002012,0.00004325217,0.0000013550925,0.000021866601,0.000095661344,0.00010041241,0.00065964734],"genre_scores_gemma":[0.9882599,0.000094754025,0.010579662,0.000018203391,0.000001686493,0.000026435562,0.00052276463,0.000040682837,0.0004559106],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99808407,0.00038756552,0.00018443137,0.00048007592,0.000611894,0.0002519671],"domain_scores_gemma":[0.9846704,0.007332593,0.0029535121,0.0016020385,0.0025848525,0.0008564931],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030378092,0.00028140153,0.00030859953,0.0041869413,0.001939734,0.0016516667,0.0007148413,0.0005890493,0.00077995],"category_scores_gemma":[0.01845285,0.00037946395,0.0006514535,0.0031035482,0.0015411298,0.0028908632,0.0020272501,0.00048010246,0.00017705448],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030377175,0.0001540282,0.8839244,0.00018528033,0.00011542627,0.0012966763,0.012832746,0.0046635033,0.0074269553,0.003683659,0.00093123456,0.084482305],"study_design_scores_gemma":[0.000053179414,0.00030180643,0.8782199,0.00022610964,0.00021181966,0.0028084745,0.018792707,0.064009465,0.008876556,0.0122969095,0.014070202,0.00013279135],"about_ca_topic_score_codex":0.009343545,"about_ca_topic_score_gemma":0.013858428,"teacher_disagreement_score":0.009343545,"about_ca_system_score_codex":0.0016396794,"about_ca_system_score_gemma":0.0013360921,"threshold_uncertainty_score":0.018578291},"labels":[],"label_agreement":null},{"id":"W4220690511","doi":"10.1007/s10664-021-10109-y","title":"Why secret detection tools are not enough: It’s not just about false positives - An industrial case study","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"National Security Agency; North Carolina State University; National Science Foundation","keywords":"False positive paradox; Computer security; Computer science; True positive rate; Data science; Artificial intelligence","score_opus":0.08079733134382909,"score_gpt":0.315110970198614,"score_spread":0.2343136388547849,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220690511","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9552059,0.00073599274,0.020805717,0.003567817,0.00005040589,0.00027089324,0.00028602016,0.00022655725,0.018850636],"genre_scores_gemma":[0.9847152,0.00030967486,0.011607129,0.00030946618,0.000020008294,0.0000602753,0.00014541962,0.00006521082,0.0027676537],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9933333,0.0031403692,0.0002815986,0.0005165879,0.0022385065,0.0004896547],"domain_scores_gemma":[0.91158456,0.07458103,0.0032726403,0.0040956503,0.005498121,0.0009679065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067751654,0.0006166042,0.00028484038,0.0019874424,0.0023668895,0.0022321322,0.0014133784,0.0026756348,0.0037106834],"category_scores_gemma":[0.038387354,0.00043607302,0.00046150203,0.0014308875,0.0024916027,0.003074199,0.0016249646,0.0020168973,0.00079700106],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017271755,0.0107337115,0.4185456,0.0014890434,0.00029476525,0.031423442,0.03053616,0.030405564,0.014513139,0.04595721,0.027263237,0.38711095],"study_design_scores_gemma":[0.0016261876,0.0099753775,0.26636574,0.0026876403,0.000942971,0.075048834,0.08027659,0.24220206,0.10200413,0.084532656,0.13385545,0.00048229675],"about_ca_topic_score_codex":0.0035685364,"about_ca_topic_score_gemma":0.0072449115,"teacher_disagreement_score":0.0067751654,"about_ca_system_score_codex":0.0016506929,"about_ca_system_score_gemma":0.0014682799,"threshold_uncertainty_score":0.035830975},"labels":[],"label_agreement":null},{"id":"W4224436806","doi":"10.1007/s10664-022-10133-6","title":"Revisiting reopened bugs in open source software systems","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); Queen's University","funders":"","keywords":"Software bug; Rework; Documentation; Pipeline (software); Software; Open source; Computer science; Software engineering; Engineering; Computer security; Operating system; Embedded system","score_opus":0.03242953496458023,"score_gpt":0.29263426994624303,"score_spread":0.2602047349816628,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4224436806","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9741431,0.0015901771,0.012009648,0.0025499554,0.000165854,0.000056930872,0.00026706932,0.00021745198,0.008999792],"genre_scores_gemma":[0.992845,0.00038086317,0.0047133244,0.0002024258,0.000060458347,0.000015968708,0.00024273543,0.0001510893,0.0013880816],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99200296,0.0022353516,0.0007930174,0.0011277463,0.0033580633,0.00048282839],"domain_scores_gemma":[0.68808305,0.21334286,0.03897568,0.022586934,0.034412015,0.002599438],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009264235,0.0004147838,0.00039511028,0.0055109905,0.0013508655,0.003540713,0.0015731854,0.001430043,0.0048830113],"category_scores_gemma":[0.23523413,0.000434174,0.0005092422,0.0034405359,0.0030091729,0.00903983,0.0027172035,0.002719958,0.00043021125],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006459558,0.00095324515,0.5711127,0.0014119537,0.0002521319,0.0020487183,0.03538069,0.004958433,0.007682009,0.04702421,0.007210461,0.32131952],"study_design_scores_gemma":[0.00011401839,0.0009318228,0.77583593,0.0023781827,0.0004984571,0.0024397091,0.036117755,0.038345963,0.010633381,0.095438175,0.03705132,0.00021527392],"about_ca_topic_score_codex":0.010783494,"about_ca_topic_score_gemma":0.016110009,"teacher_disagreement_score":0.010783494,"about_ca_system_score_codex":0.0020275523,"about_ca_system_score_gemma":0.0028598513,"threshold_uncertainty_score":0.04899454},"labels":[],"label_agreement":null},{"id":"W4225714444","doi":"10.1007/s10664-022-10125-6","title":"Tracking bad updates in mobile apps: a search-based approach","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thompson Rivers University; Queen's University; École de Technologie Supérieure; Université du Québec à Montréal","funders":"","keywords":"Android (operating system); Computer science; Benchmark (surveying); Genetic programming; Machine learning; Sorting; Mobile apps; Artificial intelligence; World Wide Web","score_opus":0.0278380218906817,"score_gpt":0.2811678476185983,"score_spread":0.2533298257279166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225714444","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7395105,0.013187367,0.20575517,0.0031977985,0.00046554412,0.001448154,0.011468517,0.0049152435,0.02005182],"genre_scores_gemma":[0.9396834,0.0011799585,0.04982265,0.00037548953,0.00023035625,0.00017012689,0.0034920042,0.00013056304,0.0049154647],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9928349,0.0013120369,0.0007043661,0.0015074832,0.0031620692,0.00047909844],"domain_scores_gemma":[0.96330196,0.024020134,0.004061481,0.0026193971,0.0052065547,0.0007904511],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041549,0.0014974352,0.0029452988,0.017952418,0.001364293,0.004374091,0.0032952449,0.004001861,0.0031607132],"category_scores_gemma":[0.035126135,0.0007482301,0.0014588556,0.010671931,0.0009336322,0.0067162104,0.0025215358,0.0016855194,0.0018730769],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022371826,0.0025184357,0.35848644,0.0020611829,0.0012866027,0.0014559028,0.0018356026,0.029350925,0.011510709,0.0103651285,0.020777637,0.55811423],"study_design_scores_gemma":[0.00017648835,0.0016616689,0.14718822,0.00042277836,0.0014617081,0.003649278,0.0029411863,0.79243416,0.012623606,0.02668721,0.01045599,0.0002977666],"about_ca_topic_score_codex":0.012468879,"about_ca_topic_score_gemma":0.02214521,"teacher_disagreement_score":0.017952418,"about_ca_system_score_codex":0.0010579374,"about_ca_system_score_gemma":0.0017120145,"threshold_uncertainty_score":0.024792612},"labels":[],"label_agreement":null},{"id":"W4226394169","doi":"10.1007/s10664-022-10139-0","title":"Studying logging practice in test code","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Concordia University","funders":"","keywords":"Logging; Test (biology); Production (economics); Computer science; Software; Database; Operating system; Forestry; Geography; Ecology","score_opus":0.01950508164430709,"score_gpt":0.27719995302169476,"score_spread":0.25769487137738767,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226394169","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.994865,0.0001330738,0.0028047913,0.00025760842,0.0000051946895,0.00001897404,0.000029123523,0.000040287396,0.0018459907],"genre_scores_gemma":[0.99841404,0.000058519563,0.0008579539,0.00003350307,0.000004081564,0.000013452684,0.000034293156,0.0000204293,0.00056368735],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98958594,0.005998087,0.0006020627,0.00083267485,0.0023750851,0.0006061375],"domain_scores_gemma":[0.59364563,0.3149517,0.046423357,0.019583728,0.02053475,0.004860897],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010134223,0.00033837988,0.00025600643,0.0024236734,0.0010705349,0.0021411066,0.001498757,0.0013433773,0.0025549803],"category_scores_gemma":[0.2138219,0.00050992024,0.00020074828,0.0021858248,0.0017988492,0.0049436935,0.0016015597,0.0018816503,0.00054379663],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004117781,0.002057551,0.850424,0.0001599992,0.000089927395,0.0003785547,0.021305352,0.003282125,0.0033749382,0.0042231493,0.00096813735,0.11332447],"study_design_scores_gemma":[0.000068669564,0.0023337747,0.9069876,0.00044853688,0.000106575,0.0010155113,0.03719853,0.028638087,0.009067578,0.0084555,0.0055748345,0.00010473775],"about_ca_topic_score_codex":0.005425374,"about_ca_topic_score_gemma":0.011579426,"teacher_disagreement_score":0.010134223,"about_ca_system_score_codex":0.0019765415,"about_ca_system_score_gemma":0.002095123,"threshold_uncertainty_score":0.053595543},"labels":[],"label_agreement":null},{"id":"W4235836287","doi":"10.1007/s10664-006-9035-z","title":"In this issue","year":2007,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science","score_opus":0.016852074762018047,"score_gpt":0.29734895763221014,"score_spread":0.2804968828701921,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4235836287","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00097618316,0.0097836405,0.001645003,0.32010975,0.49115855,0.00008554225,0.0006687955,0.0005434198,0.17502911],"genre_scores_gemma":[0.0058488743,0.006069627,0.0008858795,0.12181053,0.2075441,0.00008476738,0.0009442909,0.0004407938,0.6563712],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980057,0.0003175965,0.000119821896,0.00028819204,0.0009749062,0.00029385873],"domain_scores_gemma":[0.98794776,0.0036525596,0.0005267695,0.0015221138,0.003526355,0.0028244287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034078602,0.00088284654,0.0010409387,0.0026782164,0.0037070822,0.010140793,0.0023322394,0.00798006,0.20432112],"category_scores_gemma":[0.018347368,0.00051678554,0.0010215071,0.0016036349,0.0022222013,0.0045187455,0.0026210682,0.0093073705,0.09292948],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000124192675,0.000040918618,0.00034020422,0.00004168653,0.0000040101454,0.000024999767,0.000029855073,0.0000091789725,0.00005542639,0.0026903849,0.98078305,0.015967878],"study_design_scores_gemma":[0.000010385282,0.000010188832,0.00080602453,0.00009245917,0.000008389661,0.00003992362,0.0001332052,0.000045174766,0.00010063653,0.002758734,0.99598795,0.0000068538966],"about_ca_topic_score_codex":0.0026252763,"about_ca_topic_score_gemma":0.010801178,"teacher_disagreement_score":0.20432112,"about_ca_system_score_codex":0.0016735508,"about_ca_system_score_gemma":0.003731442,"threshold_uncertainty_score":0.6835222},"labels":[],"label_agreement":null},{"id":"W4235970556","doi":"10.1007/s10664-021-10009-1","title":"Conclusion stability for natural language based mining of design discussions","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Artifact (error); Code refactoring; Computer science; Relevance (law); Task (project management); Context (archaeology); Documentation; Stability (learning theory); Software; Machine learning; Artificial intelligence; Data mining; Software engineering; Engineering; Programming language; Systems engineering","score_opus":0.04044770805115119,"score_gpt":0.3104229520862787,"score_spread":0.2699752440351275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4235970556","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.114228204,0.0006064289,0.8683203,0.0015344034,0.00013686516,0.0003135786,0.0048999796,0.0043619056,0.0055983276],"genre_scores_gemma":[0.81178266,0.00024903897,0.17403032,0.0004074845,0.00018219544,0.00046872115,0.007869638,0.0007573651,0.0042525823],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98853904,0.0031442377,0.0010264288,0.0034957202,0.0031161695,0.0006784639],"domain_scores_gemma":[0.9044801,0.06735961,0.004227835,0.010062193,0.012591362,0.0012788267],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.009759915,0.0008006552,0.0012231056,0.0061818943,0.002081123,0.0045421235,0.0022697428,0.0013851052,0.010729092],"category_scores_gemma":[0.094172716,0.00063927524,0.0021078405,0.0030748595,0.0019244993,0.0072639263,0.0033878458,0.002233748,0.003211234],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033898847,0.0008010621,0.08377755,0.0017141671,0.00053238875,0.0007561265,0.0048597464,0.08771267,0.0347743,0.24976005,0.029687168,0.5022349],"study_design_scores_gemma":[0.00009342159,0.00024388322,0.009610051,0.0001787125,0.0001502574,0.0003690852,0.0011551714,0.7075464,0.023066888,0.24840015,0.00910558,0.00008050787],"about_ca_topic_score_codex":0.004450572,"about_ca_topic_score_gemma":0.003429539,"teacher_disagreement_score":0.9902401,"about_ca_system_score_codex":0.002320975,"about_ca_system_score_gemma":0.002666763,"threshold_uncertainty_score":0.051616013},"labels":[],"label_agreement":null},{"id":"W4240721973","doi":"10.1007/s10664-020-09811-0","title":"Guest Editorial: Special Issue on Predictive Models and Data Analytics in Software Engineering","year":2020,"lang":"en","type":"editorial","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Predictive analytics; Data science; Analytics; Data analysis; Software analytics; Software engineering; Software; Data mining; Software development; Software development process; Programming language","score_opus":0.027203129250548533,"score_gpt":0.2815026032040564,"score_spread":0.2542994739535079,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4240721973","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000021709146,0.0035061068,0.0001667993,0.017394375,0.9774695,0.000013180478,0.000060808787,0.00004195716,0.0013256032],"genre_scores_gemma":[0.00021140138,0.0019875958,0.00010331769,0.006371015,0.9845622,0.000014215026,0.000036863796,0.000041069197,0.0066722794],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9926807,0.0014437378,0.00074840867,0.0009412423,0.0038061554,0.00037973255],"domain_scores_gemma":[0.95792186,0.018650604,0.0024970942,0.00097252993,0.015078987,0.0048789033],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009804362,0.0046119336,0.0049316958,0.005743691,0.0031965505,0.012286167,0.0031082446,0.013681323,0.030836603],"category_scores_gemma":[0.036919102,0.0013421451,0.0028759418,0.0023406302,0.002332863,0.0048520085,0.0017804726,0.016176753,0.018428681],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000029530729,0.000013393361,0.000018532346,0.00014476711,0.000016366892,0.000048496087,0.000003858178,0.00002601474,0.00003466957,0.00020963053,0.9962566,0.003198312],"study_design_scores_gemma":[0.00012925938,0.00005003012,0.00042211128,0.00056778156,0.00009255759,0.00021930257,0.000036062294,0.00061833364,0.00014484389,0.0031066143,0.9945809,0.000032129334],"about_ca_topic_score_codex":0.0014105695,"about_ca_topic_score_gemma":0.005129549,"teacher_disagreement_score":0.030836603,"about_ca_system_score_codex":0.0028884504,"about_ca_system_score_gemma":0.0034586492,"threshold_uncertainty_score":0.10315877},"labels":[],"label_agreement":null},{"id":"W4250654813","doi":"10.1007/s10664-007-9041-9","title":"In this issue","year":2007,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science","score_opus":0.018849793462910394,"score_gpt":0.30064744145276495,"score_spread":0.28179764798985457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4250654813","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0007511217,0.009938868,0.0012786861,0.22083767,0.58960927,0.000089364075,0.00048288712,0.000465004,0.1765471],"genre_scores_gemma":[0.0039857044,0.00520246,0.00061968266,0.09185476,0.21899368,0.000070594484,0.00057668635,0.0003167605,0.67837965],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983438,0.0002290861,0.0000940467,0.00023383199,0.0008493219,0.00025001125],"domain_scores_gemma":[0.99286026,0.0017164533,0.0003938139,0.00080279866,0.002136507,0.00209021],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002574584,0.0008418999,0.000912213,0.0022071104,0.0033259508,0.009660361,0.0021291634,0.0074923863,0.21002243],"category_scores_gemma":[0.011808608,0.00047851843,0.0009396828,0.0014096851,0.0017252048,0.0038372288,0.002521483,0.0075734593,0.10152897],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000014415005,0.000035307017,0.00023460817,0.00004798298,0.0000036725064,0.000031026768,0.000028428512,0.000008417489,0.00008082844,0.0018891385,0.9798364,0.017789746],"study_design_scores_gemma":[0.000007983311,0.00000883896,0.0004318654,0.000058596306,0.0000053522017,0.00003732751,0.00008188314,0.00002585187,0.000081058555,0.0013601817,0.99789643,0.0000045912857],"about_ca_topic_score_codex":0.001977279,"about_ca_topic_score_gemma":0.008431475,"teacher_disagreement_score":0.78997755,"about_ca_system_score_codex":0.0014676962,"about_ca_system_score_gemma":0.0033502246,"threshold_uncertainty_score":0.702595},"labels":[],"label_agreement":null},{"id":"W4252966483","doi":"10.1007/s10664-021-10000-w","title":"FACER: An API usage-based code-example recommender for opportunistic reuse","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Snippet; Java; Application programming interface; Code reuse; Android (operating system); Source code; Cluster analysis; Reuse; Information retrieval; Code (set theory); Software; World Wide Web; Data mining; Programming language; Operating system; Artificial intelligence; Set (abstract data type)","score_opus":0.11146040033542123,"score_gpt":0.3362907313551917,"score_spread":0.22483033101977049,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4252966483","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32363805,0.0031377904,0.5106342,0.0016845439,0.00041722297,0.00087369734,0.012865813,0.11359651,0.033152174],"genre_scores_gemma":[0.5310368,0.00066317356,0.41542402,0.00068959605,0.00010547187,0.0003611962,0.01875448,0.001911475,0.031053754],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985934,0.0003388387,0.00007796582,0.0002626553,0.0006376084,0.00008959308],"domain_scores_gemma":[0.99485236,0.0020717331,0.00024818906,0.0015285783,0.0010155382,0.00028351892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012389029,0.0008834961,0.00094141864,0.0023194058,0.0006383525,0.0009772158,0.0018480283,0.0015010866,0.0062334975],"category_scores_gemma":[0.0104145985,0.0004253397,0.00076001993,0.0014398305,0.00020956516,0.0027800235,0.0014933329,0.0011132848,0.004971926],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009704362,0.001717135,0.04382422,0.0007863286,0.00040104034,0.00060865586,0.00062955497,0.011081082,0.015197176,0.0041123354,0.13221872,0.7884534],"study_design_scores_gemma":[0.00030861018,0.0007979674,0.023961853,0.00016096818,0.0003165022,0.0013117002,0.000539704,0.84688896,0.0185285,0.00879581,0.09814468,0.00024478493],"about_ca_topic_score_codex":0.012076695,"about_ca_topic_score_gemma":0.051054098,"teacher_disagreement_score":0.012076695,"about_ca_system_score_codex":0.00047992286,"about_ca_system_score_gemma":0.0009638646,"threshold_uncertainty_score":0.024012804},"labels":[],"label_agreement":null},{"id":"W4281476990","doi":"10.1007/s10664-022-10181-y","title":"Understanding and supporting the design systems practice","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Design education; Design brief; Computer science; Design knowledge; Systems design; Knowledge management; Design technology; User experience design; User interface; Systems engineering; Design review (U.S. government); Human–computer interaction; Process management; Engineering; Software engineering","score_opus":0.11921303592575368,"score_gpt":0.30020229743708404,"score_spread":0.18098926151133038,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281476990","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2077253,0.005480206,0.48794806,0.08159331,0.00026555444,0.0002904335,0.00016713986,0.0005109332,0.21601906],"genre_scores_gemma":[0.93556994,0.0017510586,0.05826907,0.0008351966,0.0000988694,0.00012458682,0.00008228484,0.0000772134,0.0031917884],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.97204393,0.019619996,0.0009762546,0.0018178994,0.004731101,0.0008107571],"domain_scores_gemma":[0.82348275,0.14783679,0.0059962105,0.015316695,0.005877977,0.0014896397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025129193,0.00087065686,0.00066287525,0.004474497,0.0027509115,0.014474954,0.0019907618,0.0051866807,0.007253137],"category_scores_gemma":[0.07828876,0.00079593237,0.00053225644,0.0026269797,0.026892534,0.023318376,0.0050950227,0.0047310134,0.0011167773],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000021752661,0.0001665135,0.0058552246,0.0002988187,0.000023332634,0.00008356503,0.010382634,0.002160348,0.00074897247,0.93470854,0.0013294765,0.044220872],"study_design_scores_gemma":[0.000024190598,0.000042875894,0.002492134,0.00034214696,0.000018122413,0.000076547614,0.0064391424,0.005490332,0.0011101532,0.9520912,0.03185554,0.000017709684],"about_ca_topic_score_codex":0.0039262273,"about_ca_topic_score_gemma":0.0030491422,"teacher_disagreement_score":0.025129193,"about_ca_system_score_codex":0.0053930962,"about_ca_system_score_gemma":0.013070522,"threshold_uncertainty_score":0.1328975},"labels":[],"label_agreement":null},{"id":"W4281617421","doi":"10.1007/s10664-021-10108-z","title":"Evolving software system families in space and time with feature revisions","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Bundesministerium für Digitalisierung und Wirtschaftsstandort; Österreichische Forschungsförderungsgesellschaft; Fundação Carlos Chagas Filho de Amparo à Pesquisa do Estado do Rio de Janeiro; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Austrian Science Fund; Österreichische Nationalstiftung für Forschung, Technologie und Entwicklung","keywords":"Feature (linguistics); Correctness; Software; Computer science; Precision and recall; Feature vector; Software system; Feature model; Data mining; Space (punctuation); Artificial intelligence; Algorithm; Programming language","score_opus":0.00914075863851467,"score_gpt":0.23168815875263116,"score_spread":0.2225474001141165,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281617421","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.56142074,0.00079241325,0.42325243,0.0002566984,0.000056782286,0.00036736001,0.000774236,0.011048014,0.0020312793],"genre_scores_gemma":[0.63469577,0.00028326648,0.36118582,0.000054361288,0.000023584354,0.00009047897,0.0019257631,0.0005991124,0.0011418973],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9930528,0.0017816155,0.0006627378,0.0014028754,0.002905462,0.00019460723],"domain_scores_gemma":[0.96519667,0.016927266,0.005217644,0.0063546333,0.00574698,0.0005567843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038119457,0.000853244,0.0005946675,0.0043601627,0.00060713745,0.0017951397,0.0009901427,0.0007779566,0.00073396444],"category_scores_gemma":[0.03484239,0.0005949187,0.000818731,0.00284178,0.00065088324,0.0028681417,0.0016223914,0.0008460174,0.0005288979],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060662476,0.00028963556,0.11481662,0.0006803049,0.00020303305,0.0009989283,0.0034148917,0.039642412,0.04765498,0.003194096,0.0029826201,0.7855158],"study_design_scores_gemma":[0.00007993392,0.00077000656,0.08710884,0.00031270244,0.00030904874,0.004517813,0.0017732854,0.7240062,0.13777155,0.011868037,0.031296607,0.00018600834],"about_ca_topic_score_codex":0.0020620045,"about_ca_topic_score_gemma":0.0028306947,"teacher_disagreement_score":0.0043601627,"about_ca_system_score_codex":0.00067963434,"about_ca_system_score_gemma":0.00069395587,"threshold_uncertainty_score":0.020159721},"labels":[],"label_agreement":null},{"id":"W4281627483","doi":"10.1007/s10664-022-10156-z","title":"A qualitative study of developers’ discussions of their problems and joys during the early COVID-19 months","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Trent University; University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada; University of Calgary","keywords":"Notice; Preparedness; Documentation; Coronavirus disease 2019 (COVID-19); Loneliness; Work (physics); Software; Psychology; Medical education; Computer science; Public relations; World Wide Web; Engineering; Management; Political science; Medicine; Social psychology","score_opus":0.05312803563303627,"score_gpt":0.33345928796693347,"score_spread":0.2803312523338972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281627483","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98847395,0.00031206643,0.0018182003,0.0018178007,0.00010455578,0.00022078381,0.00030296633,0.0000426601,0.0069070854],"genre_scores_gemma":[0.9925041,0.0002777413,0.0009449667,0.0010604017,0.000038862563,0.0005072454,0.00016600655,0.000074402626,0.0044263136],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9867154,0.008361846,0.0004907317,0.00083437905,0.0018335197,0.0017641281],"domain_scores_gemma":[0.87172925,0.09271057,0.008748489,0.0023069507,0.013740353,0.010764371],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016323544,0.0007262795,0.00083857373,0.0031215185,0.011074431,0.005065698,0.0021400596,0.002968,0.0035147883],"category_scores_gemma":[0.07520336,0.0011172806,0.0003441955,0.0025861901,0.006906516,0.0037853802,0.006858384,0.004808002,0.00074535],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006104075,0.00007769174,0.0064641996,0.00007565801,0.0000029302466,0.00048769565,0.9882472,0.000016800574,0.00081892504,0.00040344172,0.000563664,0.0027808384],"study_design_scores_gemma":[0.000011157686,0.00010583705,0.011407321,0.00016083909,0.0000038490934,0.000139158,0.97934556,0.000057439913,0.00037130498,0.00017197916,0.008200776,0.000024676014],"about_ca_topic_score_codex":0.02379125,"about_ca_topic_score_gemma":0.054063085,"teacher_disagreement_score":0.02379125,"about_ca_system_score_codex":0.006378056,"about_ca_system_score_gemma":0.0066298232,"threshold_uncertainty_score":0.08632821},"labels":[],"label_agreement":null},{"id":"W4281811500","doi":"10.1007/s10664-022-10153-2","title":"Works for Me! Cannot Reproduce – A Large Scale Empirical Study of Non-reproducible Bugs","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Dalhousie University","funders":"","keywords":"Software bug; Computer science; Empirical research; Software; Debugging; Eclipse; Software regression; Software engineering; Data science; Software development; Software quality; Programming language; Statistics","score_opus":0.030991629615728825,"score_gpt":0.32073289073010847,"score_spread":0.28974126111437964,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281811500","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94520026,0.0016153945,0.014062216,0.015152443,0.00041886457,0.0001827723,0.0013145094,0.0005859067,0.021467648],"genre_scores_gemma":[0.98853195,0.00031438732,0.0031277437,0.002137383,0.00011123024,0.00009122654,0.00066942314,0.00043519217,0.0045815124],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9907906,0.003922503,0.00055563624,0.0014555434,0.0029263783,0.00034935607],"domain_scores_gemma":[0.7262294,0.14398326,0.032816518,0.075453974,0.016830187,0.0046865777],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.013277922,0.0005016173,0.0005506806,0.001769731,0.0020028243,0.002547783,0.0018459433,0.0025598893,0.010098525],"category_scores_gemma":[0.18588398,0.00069071294,0.000652021,0.0020760836,0.0044168807,0.0062677665,0.00256245,0.002904385,0.004457865],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010181455,0.0016604488,0.6395112,0.001340607,0.0009228447,0.0015750512,0.020517819,0.0030566845,0.0056141415,0.04858618,0.05541445,0.22078247],"study_design_scores_gemma":[0.0005155413,0.0026547755,0.7123674,0.0016379893,0.0005348653,0.0040498525,0.023436608,0.009988741,0.005254822,0.11931348,0.1198994,0.000346518],"about_ca_topic_score_codex":0.0021744922,"about_ca_topic_score_gemma":0.0028116996,"teacher_disagreement_score":0.98672205,"about_ca_system_score_codex":0.00079553167,"about_ca_system_score_gemma":0.001560508,"threshold_uncertainty_score":0.070221186},"labels":[],"label_agreement":null},{"id":"W4283068924","doi":"10.1007/s10664-022-10167-w","title":"A mixed-methods analysis of micro-collaborative coding practices in OpenStack","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Fonds De La Recherche Scientifique - FNRS","keywords":"Commit; Thematic analysis; Computer science; Data science; Coding (social sciences); Knowledge management; Empirical research; Qualitative research; World Wide Web; Database","score_opus":0.04646674099401797,"score_gpt":0.3676935198609035,"score_spread":0.3212267788668855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283068924","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.93770176,0.00020389746,0.056233183,0.00014770064,0.00003530568,0.0019926243,0.00073915394,0.00013859547,0.0028077576],"genre_scores_gemma":[0.8891373,0.00014685838,0.09888535,0.00018306024,0.000025739924,0.0065007624,0.0013425495,0.00025800223,0.003520248],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9474697,0.037846494,0.0029319318,0.0036745933,0.006791962,0.0012853079],"domain_scores_gemma":[0.6303146,0.28453088,0.01770255,0.021826813,0.043431092,0.002194129],"candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.03497263,0.000507384,0.00067701965,0.004154407,0.0027230137,0.003496494,0.0020113133,0.00094634504,0.0035390917],"category_scores_gemma":[0.17204122,0.0006248102,0.0010755562,0.004241102,0.001884426,0.0030786756,0.0042226217,0.0013406518,0.00072819646],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003271622,0.0034967894,0.19862986,0.0028195493,0.0009928383,0.000516388,0.36840594,0.0045552156,0.011787893,0.019018328,0.003249318,0.38325638],"study_design_scores_gemma":[0.0012085774,0.0063171173,0.50420856,0.003438677,0.0016696431,0.0010558141,0.3169172,0.058685552,0.038712975,0.028616104,0.0384133,0.00075640326],"about_ca_topic_score_codex":0.006179458,"about_ca_topic_score_gemma":0.01116242,"teacher_disagreement_score":0.99727696,"about_ca_system_score_codex":0.004023707,"about_ca_system_score_gemma":0.008016061,"threshold_uncertainty_score":0.18495512},"labels":[],"label_agreement":null},{"id":"W4284882237","doi":"10.1007/s10664-022-10119-4","title":"FIXME: synchronize with database! An empirical study of data access self-admitted technical debt","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Technical debt; Code refactoring; Computer science; Commit; Maintainability; Database; Data access; Software development; Software; Data science; Software engineering; Operating system","score_opus":0.06157596337977576,"score_gpt":0.35714592806012,"score_spread":0.29556996468034424,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4284882237","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99648595,0.00009784149,0.0006240519,0.0004572428,0.000009590053,0.000017964107,0.00025080692,0.00007136282,0.001985087],"genre_scores_gemma":[0.9982602,0.00004582706,0.00039132594,0.00010457865,0.000013277841,0.000013426953,0.00030207002,0.00003849149,0.0008308],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99709034,0.0015764888,0.00022589105,0.0002964195,0.0005733564,0.00023747118],"domain_scores_gemma":[0.85123694,0.106118105,0.023534797,0.011154841,0.0039941724,0.003961211],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060131955,0.00022423542,0.00026957784,0.0013265477,0.0011179487,0.002298464,0.0012023099,0.0015309979,0.0048948172],"category_scores_gemma":[0.09435685,0.00041081605,0.00018526743,0.0019960941,0.0016679452,0.0056459066,0.0021968987,0.002964011,0.000994911],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094149954,0.0021947755,0.92972374,0.00015411396,0.0000863527,0.0006224796,0.010702951,0.0014661413,0.0012897394,0.01252768,0.0073598027,0.032930795],"study_design_scores_gemma":[0.00028321156,0.0020269917,0.9014994,0.0002347071,0.00014911253,0.0016958785,0.03178142,0.023467032,0.0038039563,0.01227859,0.02265626,0.00012345357],"about_ca_topic_score_codex":0.0052152895,"about_ca_topic_score_gemma":0.005116061,"teacher_disagreement_score":0.0060131955,"about_ca_system_score_codex":0.00077181193,"about_ca_system_score_gemma":0.0008004588,"threshold_uncertainty_score":0.031801224},"labels":[],"label_agreement":null},{"id":"W4286697220","doi":"10.1007/s10664-022-10157-y","title":"Predicting sensitive information leakage in IoT applications using flows-aware machine learning approach","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Source code; Process (computing); Information leakage; Control flow; Machine learning; Code (set theory); Data mining; Taint checking; Vulnerability (computing); Artificial intelligence; Statement (logic); Computer security; Software; Programming language","score_opus":0.013390852009675411,"score_gpt":0.24466050194561964,"score_spread":0.23126964993594423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4286697220","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.80054075,0.0011000886,0.19322546,0.00050809205,0.00009924274,0.00009962033,0.0006228582,0.0015836011,0.0022202432],"genre_scores_gemma":[0.9917297,0.00014936159,0.0073904386,0.000037592366,0.000029053786,0.00001376949,0.00021047458,0.000015190719,0.00042441438],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99952805,0.00009823157,0.000033629876,0.000103192426,0.00015242172,0.000084571315],"domain_scores_gemma":[0.99681866,0.0019148316,0.00052019226,0.00024226675,0.0004123927,0.00009161865],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007662216,0.0008765687,0.00044734144,0.0024317063,0.00031093552,0.0008382574,0.00040674987,0.0005839287,0.00058005826],"category_scores_gemma":[0.00490296,0.00024426472,0.0005002688,0.00090708886,0.0003404748,0.0016610948,0.0004326008,0.0009320413,0.00024481307],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088515057,0.0017151869,0.22210102,0.00029353183,0.00032889986,0.00071954756,0.00022641719,0.47687978,0.024548812,0.005047252,0.0034470758,0.26380727],"study_design_scores_gemma":[0.0000030315973,0.00006891111,0.008268364,0.000012053303,0.000025499983,0.00009257646,0.000025191677,0.9851023,0.003819233,0.0023342688,0.00024059684,0.000007960412],"about_ca_topic_score_codex":0.0018265567,"about_ca_topic_score_gemma":0.0019527235,"teacher_disagreement_score":0.0024317063,"about_ca_system_score_codex":0.00052896497,"about_ca_system_score_gemma":0.00051590457,"threshold_uncertainty_score":0.004052222},"labels":[],"label_agreement":null},{"id":"W4288079797","doi":"10.1007/s10664-020-09875-y","title":"Pandemic programming","year":2020,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Occupational Health and Safety Research","field":"Health Professions","cited_by":234,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"New York Institute of Technology; Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada; Oulun Yliopisto; Dalhousie University; University of Adelaide","keywords":"Structural equation modeling; Productivity; Confirmatory factor analysis; Notice; Pandemic; Psychology; Applied psychology; Preparedness; Human factors and ergonomics; Poison control; Computer science; Coronavirus disease 2019 (COVID-19); Medicine; Environmental health; Political science; Economics; Management; Economic growth","score_opus":0.16444328769583408,"score_gpt":0.4689126539423454,"score_spread":0.30446936624651133,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4288079797","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009144192,0.00090742507,0.0414239,0.011769002,0.004133145,0.00076682115,0.028166259,0.03503076,0.8686585],"genre_scores_gemma":[0.084910296,0.0019777706,0.05163088,0.009825068,0.0014907769,0.0010408262,0.045105524,0.010561476,0.79345745],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.99905235,0.00023636685,0.00006491843,0.0002402675,0.0002497198,0.00015637906],"domain_scores_gemma":[0.9981933,0.00045522943,0.00012833542,0.00046672113,0.00045664952,0.00029968694],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0010860193,0.00067127077,0.0003596283,0.0011084799,0.0013816647,0.0035827924,0.0016581711,0.0011177027,0.34212336],"category_scores_gemma":[0.004781862,0.0004028633,0.0007259308,0.0011409639,0.00052140286,0.0037467037,0.004148095,0.0019087337,0.14678548],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023568253,0.000106913416,0.0022144557,0.00038164225,0.0000197091,0.00036792152,0.00081626844,0.00060664664,0.001327314,0.050564833,0.75461406,0.18874459],"study_design_scores_gemma":[0.000021074751,0.000017418892,0.00058131875,0.000081879654,0.0000054674297,0.00013971368,0.00019259831,0.0005002924,0.00033203256,0.0054900413,0.9926242,0.000013844046],"about_ca_topic_score_codex":0.003747679,"about_ca_topic_score_gemma":0.0034691822,"teacher_disagreement_score":0.34212336,"about_ca_system_score_codex":0.0010123072,"about_ca_system_score_gemma":0.001863222,"threshold_uncertainty_score":0.9383812},"labels":[],"label_agreement":null},{"id":"W4288723601","doi":"10.1007/s10664-022-10158-x","title":"GBGallery : A benchmark and framework for game testing","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Benchmark (surveying); Software engineering; Video game development; Test strategy; Software development; Game testing; Software; Database; Game Developer; Game design; Artificial intelligence; Programming language; Game design document","score_opus":0.04408833245119895,"score_gpt":0.28456053815274496,"score_spread":0.240472205701546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4288723601","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013168616,0.00062440627,0.83877474,0.0009731802,0.0002669347,0.0005381795,0.0039051012,0.12838228,0.013366644],"genre_scores_gemma":[0.1539991,0.0005101744,0.8125084,0.00062160194,0.000114262846,0.0011647376,0.00998501,0.017035436,0.0040612016],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9882515,0.005074975,0.001466504,0.0008703926,0.0034686401,0.0008679527],"domain_scores_gemma":[0.96199566,0.02177425,0.0018917749,0.008415544,0.004518898,0.0014038333],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010397868,0.0025978116,0.0013873264,0.006183542,0.0009811064,0.004377573,0.007758392,0.002997756,0.009800072],"category_scores_gemma":[0.071076915,0.0012909091,0.0015327097,0.0035188543,0.0019065775,0.006404637,0.005151144,0.0043045147,0.0042933375],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012402228,0.0014061301,0.015018423,0.0018185867,0.00028010385,0.0005642026,0.0006934455,0.11018866,0.007305918,0.20291373,0.17974636,0.47882417],"study_design_scores_gemma":[0.000447403,0.0004896988,0.0034398285,0.00069471967,0.00010081218,0.000743031,0.00022608538,0.6935531,0.014915863,0.21190147,0.07330607,0.00018190895],"about_ca_topic_score_codex":0.0073888646,"about_ca_topic_score_gemma":0.007911489,"teacher_disagreement_score":0.010397868,"about_ca_system_score_codex":0.0018298962,"about_ca_system_score_gemma":0.0035579149,"threshold_uncertainty_score":0.054989815},"labels":[],"label_agreement":null},{"id":"W4289976167","doi":"10.1007/s10664-022-10168-9","title":"SSPCatcher: Learning to catch security patches","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"Institut des Sciences Mathématiques, Université du Québec à Montréal","keywords":"Computer science; Benchmarking; Computer security; Vulnerability (computing); Code (set theory); Safeguard; Source code; Programming language; Business","score_opus":0.011691828823452129,"score_gpt":0.2559203562742536,"score_spread":0.24422852745080145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4289976167","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13651446,0.0006795244,0.7532846,0.0013585201,0.0004778542,0.0003610316,0.0028142869,0.098077625,0.0064321235],"genre_scores_gemma":[0.5921382,0.00024575013,0.39120728,0.0006525572,0.0001657602,0.00032354548,0.0047497554,0.0013784354,0.009138787],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99919146,0.00022468773,0.000041173975,0.00028167645,0.00020114465,0.000059820548],"domain_scores_gemma":[0.99441564,0.0036971918,0.00023583187,0.00090627,0.0005508692,0.00019419604],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019688574,0.0012356703,0.00057830167,0.0018569265,0.00041709666,0.0008058471,0.0016514363,0.0017184084,0.0065768613],"category_scores_gemma":[0.011848587,0.0005272856,0.000710602,0.0007475309,0.0005268674,0.0022847045,0.001427186,0.002090754,0.003347452],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004295134,0.00083987357,0.03501493,0.00026600398,0.00025580838,0.00017863415,0.00013624612,0.058194026,0.0067011802,0.004212429,0.06746131,0.82631004],"study_design_scores_gemma":[0.00008577755,0.00021538248,0.0028634677,0.000025687048,0.000046579327,0.00015225307,0.000048165464,0.97384936,0.006672585,0.010495573,0.005525617,0.000019563062],"about_ca_topic_score_codex":0.0024973394,"about_ca_topic_score_gemma":0.007735985,"teacher_disagreement_score":0.0065768613,"about_ca_system_score_codex":0.0005182506,"about_ca_system_score_gemma":0.0015673311,"threshold_uncertainty_score":0.022001803},"labels":[],"label_agreement":null},{"id":"W4289978463","doi":"10.1007/s10664-022-10172-z","title":"Correction to: Can Offline Testing of Deep Neural Networks Replace Their Online Testing?","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Artificial neural network; Deep neural networks; Artificial intelligence; Online and offline; Machine learning; Information retrieval; Data mining; Data science; Operating system","score_opus":0.024391967307452477,"score_gpt":0.2589459812873127,"score_spread":0.23455401397986025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4289978463","genre_codex":"editorial","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00052625174,0.0005233037,0.0035534343,0.10065914,0.88661945,0.000034348483,0.0032583307,0.0020851975,0.0027406556],"genre_scores_gemma":[0.089931294,0.004460049,0.023055026,0.18497351,0.48522928,0.0005118367,0.008337678,0.005008125,0.19849318],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9941571,0.00091582356,0.0010672018,0.0011713612,0.0020232764,0.0006652762],"domain_scores_gemma":[0.8946709,0.038600516,0.004846582,0.009979479,0.047378436,0.0045241746],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005334527,0.0019196512,0.0028126917,0.0028609354,0.0030230195,0.0041291867,0.004664567,0.011214642,0.12843302],"category_scores_gemma":[0.17192732,0.0013147853,0.0016478859,0.0027470042,0.00304559,0.0032965043,0.002784527,0.011861451,0.059607904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007078413,0.000012344352,0.0001966074,0.00009852449,0.000029576606,0.00032410506,0.00005038982,0.00015110655,0.0001181132,0.0022518148,0.98635733,0.01033933],"study_design_scores_gemma":[0.0002519807,0.00007787706,0.0029617096,0.00050375477,0.0000919624,0.0016264677,0.0002473451,0.0035100107,0.0019512244,0.019457111,0.9691386,0.00018192639],"about_ca_topic_score_codex":0.008099953,"about_ca_topic_score_gemma":0.01021409,"teacher_disagreement_score":0.12843302,"about_ca_system_score_codex":0.00348728,"about_ca_system_score_gemma":0.0053290147,"threshold_uncertainty_score":0.42965126},"labels":[],"label_agreement":null},{"id":"W4292291355","doi":"10.1007/s10664-022-10193-8","title":"Revisiting the debate: Are code metrics useful for measuring maintenance effort?","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"National Science Foundation of Sri Lanka; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Software maintenance; Code (set theory); Context (archaeology); Source code; Java; Software metric; Code review; Granularity; Software engineering; Data science; Software; Data mining; Software quality; Software development; Programming language; Set (abstract data type)","score_opus":0.05036001561640007,"score_gpt":0.2812554912041638,"score_spread":0.23089547558776374,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4292291355","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016974865,0.07811363,0.0074939174,0.88067746,0.006486571,0.000025905672,0.00031926282,0.00006223463,0.009846051],"genre_scores_gemma":[0.6161676,0.073521234,0.0137269655,0.25580147,0.03554881,0.00016659226,0.0006752111,0.00041265337,0.003979485],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9389174,0.034524657,0.0038015833,0.007523403,0.013848714,0.0013842678],"domain_scores_gemma":[0.35372305,0.52483314,0.021342464,0.016375424,0.07991143,0.003814475],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.08336804,0.001614027,0.0029049418,0.00812288,0.0029399935,0.012334044,0.0074853655,0.014642189,0.007251221],"category_scores_gemma":[0.4054754,0.00074309926,0.0011164166,0.00910185,0.02100223,0.030441,0.0041377135,0.018526703,0.003474131],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009903355,0.00046556268,0.028105732,0.003816533,0.0009802658,0.00018065741,0.0069424827,0.0010858311,0.0009400254,0.31394812,0.18066372,0.4618807],"study_design_scores_gemma":[0.0004058098,0.00050684984,0.04932445,0.019637082,0.00094088493,0.000577128,0.021873537,0.0069463365,0.0020244198,0.64853215,0.24884984,0.0003814751],"about_ca_topic_score_codex":0.0092609925,"about_ca_topic_score_gemma":0.0088330535,"teacher_disagreement_score":0.91663194,"about_ca_system_score_codex":0.005062594,"about_ca_system_score_gemma":0.008644125,"threshold_uncertainty_score":0.44089758},"labels":[],"label_agreement":null},{"id":"W4296481997","doi":"10.1007/s10664-022-10187-6","title":"Sources of software development task friction","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Personal Information Management and User Behavior","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Task (project management); Computer science; Work (physics); Software; Forcing (mathematics); Software development; Human–computer interaction; Software engineering; Field (mathematics); Development environment; Systems engineering; Engineering","score_opus":0.1488232714857706,"score_gpt":0.3677832485603151,"score_spread":0.21895997707454448,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4296481997","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9418442,0.0018367398,0.017374052,0.0027191374,0.00007629999,0.00031601064,0.0015396914,0.00024531598,0.03404846],"genre_scores_gemma":[0.9965153,0.00014056948,0.0015362466,0.000080953156,0.00001846089,0.00005584396,0.00031777038,0.000054412543,0.0012804327],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98466635,0.0056394585,0.0015343335,0.0013098728,0.0053688297,0.001481212],"domain_scores_gemma":[0.713138,0.20964462,0.032007452,0.029064,0.01305233,0.0030936424],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017666312,0.0006225606,0.0011514396,0.010219969,0.002187275,0.005203305,0.0016065685,0.0017764542,0.011406293],"category_scores_gemma":[0.2150694,0.0011695977,0.00064149406,0.0071684644,0.001631938,0.00322571,0.00478467,0.0029969676,0.0011118159],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002074893,0.0020860825,0.69381684,0.00097278005,0.0004950733,0.001134695,0.019667778,0.008171675,0.003233287,0.08120406,0.00833756,0.17880532],"study_design_scores_gemma":[0.0003503571,0.0004297428,0.8259795,0.0015201004,0.00040042802,0.0016795654,0.013809786,0.05011866,0.0041704294,0.08078702,0.020516034,0.00023830062],"about_ca_topic_score_codex":0.005062085,"about_ca_topic_score_gemma":0.0040309443,"teacher_disagreement_score":0.017666312,"about_ca_system_score_codex":0.0040526786,"about_ca_system_score_gemma":0.0039060237,"threshold_uncertainty_score":0.093429506},"labels":[],"label_agreement":null},{"id":"W4297903144","doi":"10.1007/s10664-022-10221-7","title":"On the usage and development of deep learning compilers: an empirical study on TVM","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Parallel Computing and Optimization Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Compiler; Computer science; Deep learning; Artificial intelligence; Software engineering; Programming language","score_opus":0.035874689591281164,"score_gpt":0.291733287783201,"score_spread":0.25585859819191986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4297903144","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99670655,0.00019768975,0.0012959068,0.00014720549,0.000005677575,0.000018040146,0.00011958127,0.000102163365,0.0014073021],"genre_scores_gemma":[0.9954204,0.00015472104,0.002664401,0.00006746973,0.0000057804523,0.000020315352,0.00033978076,0.00007493589,0.0012521779],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9953591,0.0019617048,0.00033168856,0.0004600593,0.0015496741,0.00033780374],"domain_scores_gemma":[0.84437066,0.114837244,0.020201944,0.009071404,0.008954926,0.0025638817],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054046954,0.00025353802,0.00019840371,0.0020207844,0.00046180107,0.0013551681,0.0012495082,0.0009864866,0.0016568037],"category_scores_gemma":[0.097978145,0.00038439102,0.00030846742,0.0018670674,0.0010580603,0.002610783,0.0010544528,0.0015394314,0.00058951497],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004939673,0.0015051166,0.7953655,0.00030367868,0.00012126081,0.0004209903,0.0067897686,0.0072115427,0.0042230156,0.003851221,0.002919426,0.17679448],"study_design_scores_gemma":[0.00008132604,0.0022122646,0.81698257,0.0005556786,0.00026553692,0.002278681,0.010996207,0.1164445,0.015692096,0.006457629,0.027900804,0.0001326374],"about_ca_topic_score_codex":0.0031448056,"about_ca_topic_score_gemma":0.0046634227,"teacher_disagreement_score":0.0054046954,"about_ca_system_score_codex":0.0011505258,"about_ca_system_score_gemma":0.001499545,"threshold_uncertainty_score":0.02858311},"labels":[],"label_agreement":null},{"id":"W4300960622","doi":"10.1007/s10664-022-10223-5","title":"A controlled experiment of different code representations for learning-based program repair","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Code (set theory); Source code; Representation (politics); Artificial intelligence; Point (geometry); Syntax; Programming language; Perspective (graphical); Process (computing); Natural language processing; Set (abstract data type)","score_opus":0.03017434424875222,"score_gpt":0.33110164950214477,"score_spread":0.30092730525339256,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4300960622","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98787177,0.000117832235,0.006944164,0.00007006475,0.00013803548,0.0030472132,0.00038926088,0.00031660826,0.0011051494],"genre_scores_gemma":[0.96962625,0.00016871,0.020630563,0.00022597282,0.000074833915,0.0052091903,0.00064000697,0.0001345176,0.003289957],"study_design_codex":"nonrandomized_trial","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972651,0.00088994927,0.00044590642,0.0007578822,0.00038534252,0.00025585378],"domain_scores_gemma":[0.9516662,0.038479738,0.0025755863,0.003752738,0.001680864,0.0018448676],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036833775,0.0013417272,0.0012224143,0.00053519406,0.00059527735,0.0011023228,0.0022701393,0.0021366847,0.010339895],"category_scores_gemma":[0.03323672,0.0008761733,0.000648715,0.0003312537,0.0012499796,0.0018614321,0.0010859765,0.0019963915,0.001042833],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.29330128,0.34542656,0.005654008,0.0034596324,0.0005256359,0.00031221536,0.004958994,0.012162586,0.16482407,0.002297227,0.0028073965,0.16427042],"study_design_scores_gemma":[0.12228468,0.67823935,0.040410724,0.00040748215,0.0015525065,0.00031233297,0.0017983759,0.039444145,0.10240035,0.005772006,0.006854767,0.00052331627],"about_ca_topic_score_codex":0.0015294076,"about_ca_topic_score_gemma":0.0015119208,"teacher_disagreement_score":0.010339895,"about_ca_system_score_codex":0.0006946887,"about_ca_system_score_gemma":0.001199789,"threshold_uncertainty_score":0.034590364},"labels":[],"label_agreement":null},{"id":"W4309382233","doi":"10.1007/s10664-022-10244-0","title":"Developer discussion topics on the adoption and barriers of low code software development platforms","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Agile software development; Computer science; Software development; Personalization; Software; Software development process; Software engineering; World Wide Web; Data science; Engineering","score_opus":0.02297163341640662,"score_gpt":0.2529683407359211,"score_spread":0.2299967073195145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4309382233","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.74969435,0.0043994184,0.024970649,0.1063324,0.0046021775,0.0014973372,0.0029465847,0.00072304026,0.10483411],"genre_scores_gemma":[0.92885864,0.003002153,0.008489207,0.008487807,0.0019000736,0.0018685807,0.0014911807,0.00041137714,0.045491006],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.99484575,0.0024040297,0.0003517477,0.00043889767,0.0010369663,0.00092255557],"domain_scores_gemma":[0.91411716,0.059727855,0.0056775706,0.0016836877,0.0114447465,0.007348955],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.011925642,0.0005563123,0.00041328973,0.0030358315,0.0039628083,0.0024557938,0.0008980138,0.003031293,0.018699478],"category_scores_gemma":[0.06658877,0.00041048016,0.0006039894,0.0019513088,0.00088915817,0.0046640886,0.0042990716,0.003120569,0.0016938668],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019464245,0.0019115545,0.20854053,0.0042004897,0.00014303274,0.00256719,0.22110441,0.0013995738,0.01824305,0.036366016,0.14695407,0.35662362],"study_design_scores_gemma":[0.00020965192,0.0014403147,0.2281364,0.0033742655,0.00017661584,0.0008198894,0.17245144,0.0022525585,0.010087751,0.01528695,0.5655598,0.00020435051],"about_ca_topic_score_codex":0.0023786228,"about_ca_topic_score_gemma":0.0043911813,"teacher_disagreement_score":0.98807436,"about_ca_system_score_codex":0.003054953,"about_ca_system_score_gemma":0.004682285,"threshold_uncertainty_score":0.06306958},"labels":[],"label_agreement":null},{"id":"W4309896433","doi":"10.1007/s10664-022-10242-2","title":"Automatic prediction of rejected edits in Stack Overflow","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; University of Saskatchewan","funders":"Natural Sciences and Engineering Research Council of Canada; Canada First Research Excellence Fund; Global Institute for Water Security, University of Saskatchewan","keywords":"Computer science; Quality (philosophy); World Wide Web; Information retrieval; Software; Precision and recall; Artificial intelligence; Programming language","score_opus":0.0211596723309598,"score_gpt":0.2590476734414302,"score_spread":0.2378880011104704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4309896433","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9911809,0.00023026661,0.0048055043,0.00007143606,0.00006948076,0.0000288652,0.0016516056,0.001240678,0.0007212933],"genre_scores_gemma":[0.9900726,0.00006509769,0.0056272335,0.000028760753,0.000055634984,0.000013603354,0.0032540346,0.000114926,0.00076816085],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997964,0.00032349234,0.0002081606,0.0004908502,0.0008201687,0.00019336019],"domain_scores_gemma":[0.9576629,0.023792336,0.0074135168,0.0023055088,0.0069742906,0.0018515304],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016438977,0.00062266213,0.00049685297,0.0046891384,0.0006253858,0.0012388185,0.0010590224,0.0012077418,0.0017331204],"category_scores_gemma":[0.024771202,0.00025347542,0.00043670516,0.0015328827,0.00028594973,0.0011816488,0.00068522914,0.0010602408,0.0009816389],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021389222,0.0007142964,0.8409462,0.00035329864,0.00026722456,0.0013179334,0.0005153657,0.01021805,0.017800303,0.00072409335,0.007562227,0.117442146],"study_design_scores_gemma":[0.00010535357,0.0006823094,0.54398066,0.00010520737,0.00032823626,0.0017977469,0.0005054016,0.42224658,0.024708005,0.00156569,0.0038503634,0.00012441736],"about_ca_topic_score_codex":0.004825873,"about_ca_topic_score_gemma":0.007856971,"teacher_disagreement_score":0.004825873,"about_ca_system_score_codex":0.00035891056,"about_ca_system_score_gemma":0.0009512057,"threshold_uncertainty_score":0.009595513},"labels":[],"label_agreement":null},{"id":"W4312118684","doi":"10.1007/s10664-022-10246-y","title":"How programmers find online learning resources","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Personalization; Resource (disambiguation); World Wide Web; Online learning; Knowledge management","score_opus":0.020907684315663436,"score_gpt":0.251643855097324,"score_spread":0.2307361707816606,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312118684","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96516764,0.0003361583,0.0020782675,0.001464874,0.000021081225,0.000021808051,0.00013662632,0.000057718036,0.030715786],"genre_scores_gemma":[0.98609734,0.00031330212,0.001598814,0.00033649744,0.000014219522,0.000016887356,0.0001803152,0.00009442864,0.011348156],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9978465,0.0010434422,0.000088251305,0.00025981432,0.00048390182,0.00027808503],"domain_scores_gemma":[0.9646166,0.028115584,0.0022776902,0.0013753441,0.0021862385,0.0014285088],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018631616,0.00021028653,0.00019218316,0.0017239504,0.0010049385,0.0058315615,0.00074166455,0.0014084063,0.016262015],"category_scores_gemma":[0.03836724,0.0003011479,0.00022713777,0.0020372015,0.000748739,0.009056821,0.0016649257,0.0011527231,0.002200559],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000750856,0.0023515036,0.5273947,0.0006063095,0.00017093452,0.0007757697,0.060336947,0.0017356751,0.005132572,0.017791089,0.019886035,0.3630677],"study_design_scores_gemma":[0.0001798974,0.00077599776,0.5585317,0.0009659442,0.0004198562,0.0015542409,0.26534542,0.016329996,0.007898753,0.04933237,0.098461784,0.00020399835],"about_ca_topic_score_codex":0.0065331403,"about_ca_topic_score_gemma":0.010764776,"teacher_disagreement_score":0.016262015,"about_ca_system_score_codex":0.00075880456,"about_ca_system_score_gemma":0.0010242416,"threshold_uncertainty_score":0.054401875},"labels":[],"label_agreement":null},{"id":"W4319079817","doi":"10.1007/s10664-022-10283-7","title":"What makes Ethereum blockchain transactions be processed fast or slow? An empirical study","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); Queen's University","funders":"","keywords":"Blockchain; Computer science; Database transaction; Smart contract; Transaction processing; Quality of service; Transaction data; Block (permutation group theory); Distributed transaction; Quality (philosophy); Computer security; Database; Computer network","score_opus":0.0366201365717742,"score_gpt":0.3119537041319782,"score_spread":0.275333567560204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319079817","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99710125,0.00013208247,0.00045927576,0.000216479,0.0000047885364,0.000034171037,0.000050561568,0.0000060518223,0.0019953293],"genre_scores_gemma":[0.99902093,0.00009959082,0.00019814578,0.00003387502,0.000009818525,0.000015663734,0.00008026522,0.000007634986,0.00053410983],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9951782,0.0020793728,0.00034659382,0.00056237774,0.001251842,0.0005815702],"domain_scores_gemma":[0.557597,0.37611106,0.042441636,0.006348662,0.012987028,0.004514662],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012328504,0.00037680898,0.00045506688,0.0022524945,0.0012561992,0.0034315523,0.0014348543,0.0016523493,0.008195511],"category_scores_gemma":[0.119786605,0.0004612043,0.0003884001,0.0023165143,0.0027274045,0.0067575495,0.0011643309,0.0025831612,0.0012418997],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021255428,0.003764178,0.9329154,0.0003064713,0.00015049246,0.0005815144,0.0075018913,0.003033567,0.0013210281,0.011245821,0.0012345365,0.035819486],"study_design_scores_gemma":[0.00028637983,0.002320284,0.886793,0.00030783666,0.00029155563,0.00089085347,0.05015163,0.032966647,0.0032595547,0.015801828,0.006811667,0.00011866611],"about_ca_topic_score_codex":0.0048293825,"about_ca_topic_score_gemma":0.004368832,"teacher_disagreement_score":0.012328504,"about_ca_system_score_codex":0.001605868,"about_ca_system_score_gemma":0.0018966681,"threshold_uncertainty_score":0.06520015},"labels":[],"label_agreement":null},{"id":"W4319083508","doi":"10.1007/s10664-022-10276-6","title":"An empirical study of text-based machine learning models for vulnerability detection","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Machine learning; Vulnerability (computing); Artificial intelligence; Context (archaeology); Empirical research; Source code; Function (biology); Construct (python library); Vulnerability assessment; Code (set theory); Data science; Computer security; Geography; Programming language","score_opus":0.059584817812828254,"score_gpt":0.3385498813244283,"score_spread":0.27896506351160005,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319083508","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97985816,0.0007246525,0.015138775,0.000889964,0.000078129735,0.000096510834,0.0008810761,0.00017491245,0.0021577803],"genre_scores_gemma":[0.99229395,0.0001983368,0.005103146,0.00013869347,0.00008455763,0.000060759718,0.001279091,0.000051042767,0.0007904144],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.992922,0.0049486435,0.00043053832,0.00067849684,0.000851566,0.00016883937],"domain_scores_gemma":[0.5946221,0.38334325,0.0081098545,0.006657025,0.0064205215,0.00084743445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0126240095,0.00085677,0.0005776352,0.002203456,0.00057917845,0.0018476757,0.0015777835,0.0017187053,0.0030322343],"category_scores_gemma":[0.14984278,0.0002856331,0.0006630123,0.0029152827,0.0007572208,0.005758723,0.0009314216,0.0023911395,0.0011882053],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004864189,0.008742703,0.52722716,0.0010906134,0.0009737028,0.00089739484,0.0032689979,0.11559635,0.004655746,0.008639983,0.013941054,0.31010222],"study_design_scores_gemma":[0.0001544081,0.0010560628,0.084546514,0.00015208121,0.00029682097,0.0005363505,0.0009737974,0.89776814,0.0026227369,0.009314727,0.002503626,0.000074750875],"about_ca_topic_score_codex":0.0040455637,"about_ca_topic_score_gemma":0.0029542572,"teacher_disagreement_score":0.0126240095,"about_ca_system_score_codex":0.0011769874,"about_ca_system_score_gemma":0.0006639501,"threshold_uncertainty_score":0.066762924},"labels":[],"label_agreement":null},{"id":"W4319228909","doi":"10.1007/s10664-023-10290-2","title":"Applying declarative analysis to industrial automotive software product line models","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of Toronto","funders":"General Motors of Canada; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Automotive industry; Software product line; Software engineering; Computer science; Software; Milestone; Product (mathematics); Programming language; Software development; Engineering","score_opus":0.1475329072300927,"score_gpt":0.34279995238639305,"score_spread":0.19526704515630033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319228909","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05757194,0.00006331123,0.9389448,0.0001222758,0.000010396426,0.000046664474,0.00012679011,0.0008931759,0.0022205813],"genre_scores_gemma":[0.6894426,0.00013083592,0.30782962,0.000076567354,0.000015770624,0.00008702595,0.00047752846,0.00022487878,0.0017151248],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987704,0.0005178329,0.00009157193,0.00012131865,0.00040789656,0.000090946785],"domain_scores_gemma":[0.99529064,0.0030685843,0.0003625314,0.00063408,0.00060499256,0.000039125847],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002300675,0.0005199537,0.0003517,0.0007272488,0.00034799695,0.0016312631,0.0012507131,0.00046903192,0.0017356293],"category_scores_gemma":[0.0083519425,0.0004992166,0.0007531762,0.0006057913,0.000559461,0.0014485333,0.00090524036,0.0010734065,0.00028535383],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010571844,0.0003701535,0.008154239,0.00022906941,0.000098944205,0.00034447917,0.00092572055,0.6995578,0.01113011,0.14263034,0.0013672588,0.13508622],"study_design_scores_gemma":[0.000008034823,0.00003459413,0.00046348516,0.000017728924,0.000019242452,0.000024567878,0.000074922835,0.9724917,0.0031095238,0.02251608,0.0012342724,0.00000582137],"about_ca_topic_score_codex":0.0061141606,"about_ca_topic_score_gemma":0.010178431,"teacher_disagreement_score":0.0061141606,"about_ca_system_score_codex":0.00086569274,"about_ca_system_score_gemma":0.0016398014,"threshold_uncertainty_score":0.012167275},"labels":[],"label_agreement":null},{"id":"W4319459392","doi":"10.1007/s10664-022-10270-y","title":"Assessing the exposure of software changes","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Deliverable; Computer science; Ranking (information retrieval); Software; Source code; Task (project management); Code (set theory); Set (abstract data type); Software engineering; Reliability engineering; Data mining; Systems engineering; Engineering; Machine learning; Operating system; Programming language","score_opus":0.050832639175595495,"score_gpt":0.3292309896640568,"score_spread":0.2783983504884613,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319459392","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99514365,0.00008467989,0.0014829638,0.00010821069,0.0000067990595,0.000029417057,0.000116245676,0.000026138683,0.0030019365],"genre_scores_gemma":[0.99840564,0.000059446735,0.0007409737,0.000026817801,0.00000926256,0.000015792066,0.00014965718,0.0000061334677,0.00058624893],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99299616,0.0024018646,0.00060785055,0.0006799315,0.0029218148,0.0003923766],"domain_scores_gemma":[0.86608607,0.08814725,0.027384745,0.0067336857,0.008198123,0.0034501904],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052902596,0.00028921932,0.00025944403,0.0027849982,0.0003688872,0.0013126407,0.0006488798,0.0012004274,0.004065148],"category_scores_gemma":[0.098083735,0.00020502097,0.0004941428,0.001461487,0.0004986171,0.002455294,0.0014711707,0.001079174,0.0005355699],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037378422,0.00089549785,0.93878716,0.000071512724,0.00017543165,0.000082653125,0.0011503345,0.001519352,0.0017695104,0.00078155217,0.00023949066,0.054153647],"study_design_scores_gemma":[0.000020144944,0.0016212511,0.9836838,0.000038703143,0.000093404225,0.0002589593,0.001727269,0.0069231945,0.002638925,0.0017559201,0.0012121939,0.000026287194],"about_ca_topic_score_codex":0.0014085177,"about_ca_topic_score_gemma":0.0016092929,"teacher_disagreement_score":0.0052902596,"about_ca_system_score_codex":0.0006388686,"about_ca_system_score_gemma":0.00058374647,"threshold_uncertainty_score":0.027977884},"labels":[],"label_agreement":null},{"id":"W4320039274","doi":"10.1007/s10664-022-10262-y","title":"Towards understanding quality challenges of the federated learning for neural networks: a first look from the lens of robustness","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Alberta; University of Calgary","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Robustness (evolution); Computer science; Machine learning; Artificial intelligence; Artificial neural network; Deep learning; Data mining; Deep neural networks; Process (computing)","score_opus":0.1321526516824001,"score_gpt":0.30885220074616543,"score_spread":0.17669954906376534,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4320039274","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039027352,0.004117801,0.9288759,0.021375706,0.00016480606,0.00006233822,0.00026296507,0.00019395912,0.0059190365],"genre_scores_gemma":[0.87280124,0.004618602,0.11680639,0.0017209729,0.0009980714,0.0001419325,0.00029886782,0.00027246372,0.0023414118],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98258466,0.0086431075,0.0010081825,0.0021299447,0.004817728,0.00081648695],"domain_scores_gemma":[0.86617315,0.092740506,0.009000159,0.02194425,0.008753353,0.0013885279],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.030413678,0.0010477658,0.002453238,0.0026239955,0.0012998619,0.010794192,0.0040198197,0.0048385486,0.0043085506],"category_scores_gemma":[0.15514421,0.0010991556,0.0016815371,0.0022181028,0.010752186,0.029827874,0.007369714,0.008074089,0.00045556662],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000099161975,0.00007110637,0.0025719728,0.00019741112,0.00011677052,0.0000705483,0.00032459912,0.047433995,0.00053590455,0.92304325,0.0014256907,0.024109626],"study_design_scores_gemma":[0.000009247159,0.000025053767,0.00039730704,0.00007131888,0.000015711745,0.000034349072,0.00010965998,0.08615491,0.00034797695,0.9115112,0.0013079095,0.00001540111],"about_ca_topic_score_codex":0.0027840515,"about_ca_topic_score_gemma":0.0013618447,"teacher_disagreement_score":0.030413678,"about_ca_system_score_codex":0.003935039,"about_ca_system_score_gemma":0.0032804327,"threshold_uncertainty_score":0.1608448},"labels":[],"label_agreement":null},{"id":"W4320882938","doi":"10.1007/s10664-022-10271-x","title":"Refactoring practices in the context of data-intensive systems","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Code refactoring; Computer science; Data access; Context (archaeology); Maintainability; Database; Software engineering; Software; Programming language","score_opus":0.16027050351680053,"score_gpt":0.3800548519220058,"score_spread":0.21978434840520525,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4320882938","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9960426,0.00058974425,0.0011032347,0.0004754391,0.000006115136,0.000019485937,0.00003445896,0.000015365506,0.0017134441],"genre_scores_gemma":[0.9980469,0.00025481687,0.0014369104,0.00003426858,0.0000037302905,0.000005742885,0.000031834057,0.000007524629,0.00017830936],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99152595,0.0050004474,0.00061073346,0.0008438673,0.00149968,0.0005192598],"domain_scores_gemma":[0.88862044,0.08249551,0.0114788655,0.00533265,0.0096119605,0.0024606124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008906294,0.0002451576,0.00018764618,0.0021357932,0.0015275613,0.0026189452,0.0010437464,0.0010454267,0.0013757874],"category_scores_gemma":[0.07565498,0.00028182066,0.0001742249,0.002892689,0.0010288983,0.0032463477,0.0016845809,0.0011693136,0.00017381342],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007624809,0.002025166,0.5039103,0.0009570632,0.000170851,0.0026367416,0.13914487,0.008526495,0.014254497,0.019302396,0.0019310232,0.30637816],"study_design_scores_gemma":[0.00014288169,0.0019853632,0.7322642,0.0015861065,0.00028675448,0.002654128,0.15440968,0.03272543,0.015359051,0.027242359,0.03113699,0.00020704539],"about_ca_topic_score_codex":0.010212273,"about_ca_topic_score_gemma":0.027542293,"teacher_disagreement_score":0.010212273,"about_ca_system_score_codex":0.0029193899,"about_ca_system_score_gemma":0.003059176,"threshold_uncertainty_score":0.047101498},"labels":[],"label_agreement":null},{"id":"W4321093054","doi":"10.1007/s10664-022-10267-7","title":"Vulnerability management in Linux distributions","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Security and Verification in Computing","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Vulnerability (computing); Vulnerability management; Computer science; Revenue; Software; Computer security; Linux kernel; Secure coding; Open source; Upstream (networking); Vulnerability assessment; Operating system; Business; Software security assurance; Telecommunications; Accounting; Information security","score_opus":0.037604282998133204,"score_gpt":0.30874347030439714,"score_spread":0.27113918730626396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4321093054","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9915036,0.00010588242,0.0062827137,0.00029984527,0.0000030262813,0.00000905893,0.000055813623,0.00009633489,0.0016437634],"genre_scores_gemma":[0.9993869,0.0000134742195,0.0004203342,0.0000041387375,0.000001438657,0.00000207333,0.000021785554,0.0000063453767,0.00014343295],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99782616,0.000983488,0.00010758514,0.00033940774,0.00041017393,0.00033323508],"domain_scores_gemma":[0.95744866,0.028249137,0.0068053124,0.0037915867,0.0026583504,0.0010469784],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042046746,0.00021649945,0.00022426622,0.0027287004,0.00081747153,0.0014996864,0.0006790548,0.0005253632,0.002101588],"category_scores_gemma":[0.054171726,0.00018984445,0.0002586384,0.0017089483,0.0013719734,0.0037372322,0.0014352738,0.00096108194,0.00014770601],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000478198,0.00031762876,0.7447213,0.00008620884,0.000114024886,0.00033496812,0.0041085244,0.058016423,0.0021816164,0.07729128,0.0023543946,0.1099954],"study_design_scores_gemma":[0.000039179668,0.00028229304,0.39364624,0.00011138791,0.00010928954,0.0008211107,0.0063135438,0.44218665,0.004216496,0.14866374,0.0035401674,0.000069963775],"about_ca_topic_score_codex":0.007222889,"about_ca_topic_score_gemma":0.0052254344,"teacher_disagreement_score":0.007222889,"about_ca_system_score_codex":0.0020710316,"about_ca_system_score_gemma":0.0011980115,"threshold_uncertainty_score":0.022236705},"labels":[],"label_agreement":null},{"id":"W4321787049","doi":"10.1007/s10664-022-10272-w","title":"Semantically-enhanced topic recommendation systems for software projects","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Metadata; Recommender system; Software; Information retrieval; World Wide Web; Software engineering; Data science; Programming language","score_opus":0.050022828276608984,"score_gpt":0.3143586557869092,"score_spread":0.2643358275103002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4321787049","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06399181,0.0018029597,0.8800558,0.0007802607,0.000418434,0.00046227706,0.0052124322,0.04284619,0.0044299015],"genre_scores_gemma":[0.2602555,0.0008125278,0.7190115,0.0002598594,0.00034441391,0.00030485925,0.012942291,0.0012438429,0.0048251706],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99725133,0.0008534176,0.00031320777,0.00064758153,0.00072767097,0.00020677855],"domain_scores_gemma":[0.9932594,0.0026879439,0.00039303498,0.001713409,0.0015878117,0.0003584253],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029212378,0.00097093027,0.0011704392,0.004982917,0.0010093796,0.0022736238,0.0014993658,0.0015157956,0.005274057],"category_scores_gemma":[0.015026502,0.00048572963,0.0015780855,0.0038787888,0.00034476363,0.0041179606,0.0030700294,0.0015259599,0.0042357426],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016184907,0.00070169894,0.012599644,0.0008899441,0.0004949311,0.00030651301,0.0012202861,0.01792742,0.027793638,0.013790015,0.04428585,0.87837154],"study_design_scores_gemma":[0.00047684423,0.00066151953,0.010202222,0.00023498763,0.00069301337,0.00071296276,0.0008571367,0.8372187,0.02743153,0.06222538,0.05906446,0.0002212998],"about_ca_topic_score_codex":0.0051426496,"about_ca_topic_score_gemma":0.013287222,"teacher_disagreement_score":0.005274057,"about_ca_system_score_codex":0.00078458217,"about_ca_system_score_gemma":0.0015026518,"threshold_uncertainty_score":0.017643452},"labels":[],"label_agreement":null},{"id":"W4323925049","doi":"10.1007/s10664-022-10277-5","title":"Registered reports in software engineering","year":2023,"lang":"en","type":"editorial","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Software engineering; Computer science; Engineering; Systems engineering","score_opus":0.027683895632145253,"score_gpt":0.2998760716857518,"score_spread":0.27219217605360657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323925049","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00004433598,0.0073862835,0.0011555975,0.0383524,0.93983257,0.000103404935,0.000117247124,0.0003423927,0.012665759],"genre_scores_gemma":[0.0014163454,0.018013798,0.0023014452,0.03740693,0.863208,0.00029015864,0.00035665085,0.0010912069,0.075915396],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.93733037,0.014283136,0.009447529,0.0035549945,0.033983506,0.0014005773],"domain_scores_gemma":[0.7015031,0.117403984,0.016086161,0.01748957,0.1319167,0.015600544],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.035975683,0.0033652834,0.0026170649,0.008670058,0.005011977,0.020694561,0.0053459504,0.017685637,0.044568412],"category_scores_gemma":[0.22360691,0.0014045018,0.0020250443,0.0066456976,0.0061459304,0.011708659,0.005214273,0.023398357,0.06401398],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000011807922,0.000008886055,0.000017169563,0.00044542196,0.0000049728924,0.00006207055,0.00006234594,0.00002614291,0.000044624805,0.0030264524,0.9843912,0.0118990075],"study_design_scores_gemma":[0.0000052858395,0.0000060311972,0.00002291184,0.00035348273,0.0000038155017,0.0000627157,0.000030398362,0.000025060443,0.00003896135,0.0010956976,0.99834895,0.0000066554453],"about_ca_topic_score_codex":0.001186232,"about_ca_topic_score_gemma":0.0021090303,"teacher_disagreement_score":0.9640243,"about_ca_system_score_codex":0.0051859817,"about_ca_system_score_gemma":0.012480186,"threshold_uncertainty_score":0.19025987},"labels":[],"label_agreement":null},{"id":"W4324374545","doi":"10.1007/s10664-022-10260-0","title":"Evaluating ensemble imputation in software effort estimation","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"","keywords":"Imputation (statistics); Computer science; Missing data; Statistics; Data mining; Regression; Mathematics; Machine learning","score_opus":0.054878849205720864,"score_gpt":0.3653485525550296,"score_spread":0.3104697033493088,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4324374545","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21975522,0.0016915279,0.77241814,0.0005957004,0.00032643866,0.00010366924,0.00081488857,0.0018583842,0.002436058],"genre_scores_gemma":[0.7629848,0.0003597117,0.23007214,0.00032172966,0.00032532163,0.00019310316,0.0034031237,0.00024276107,0.0020973256],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9901484,0.006945865,0.00042222277,0.0011519665,0.0009559924,0.00037550958],"domain_scores_gemma":[0.8882717,0.093457475,0.0022697921,0.010106132,0.0048724324,0.0010224406],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022441618,0.0009664868,0.0028661923,0.0016183393,0.0007777405,0.0015657732,0.0027468514,0.002718533,0.00253115],"category_scores_gemma":[0.08030682,0.00080683484,0.0014426069,0.0023163997,0.0005754908,0.003027127,0.002319292,0.0027768116,0.0008424649],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013491473,0.00061931554,0.045051377,0.0001794758,0.0013074942,0.00015926201,0.00020130954,0.6885693,0.0006572745,0.0071832957,0.0067350725,0.24798761],"study_design_scores_gemma":[0.00003363999,0.00010511921,0.002210315,0.000024650653,0.00006987812,0.000028324792,0.000028373594,0.9897411,0.00038840825,0.006947999,0.00041135112,0.000010835768],"about_ca_topic_score_codex":0.003716601,"about_ca_topic_score_gemma":0.004213605,"teacher_disagreement_score":0.022441618,"about_ca_system_score_codex":0.00058110745,"about_ca_system_score_gemma":0.0012490953,"threshold_uncertainty_score":0.118683994},"labels":[],"label_agreement":null},{"id":"W4360948905","doi":"10.1007/s10664-022-10278-4","title":"Empirical analysis of security vulnerabilities in Python packages","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Concordia University","funders":"","keywords":"Python (programming language); Computer science; Secure coding; Software security assurance; Software engineering; Security bug; Vulnerability management; Software; Software development; Software bug; Computer security; Ecosystem; Reusability; Programming language; Vulnerability assessment; Information security; Ecology","score_opus":0.02247784916848785,"score_gpt":0.3112819751728626,"score_spread":0.2888041260043748,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4360948905","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99800354,0.000054341544,0.0010779947,0.00007978965,0.0000013792345,0.000009279566,0.00013594556,0.0000202346,0.000617333],"genre_scores_gemma":[0.9990043,0.000034943685,0.0005206909,0.00001484686,0.0000030615033,0.000008415847,0.00017986618,0.000009450981,0.00022437029],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99748707,0.0011135201,0.00014992313,0.0003108634,0.00071082596,0.00022776764],"domain_scores_gemma":[0.86772096,0.102400415,0.016769048,0.0055826507,0.006508834,0.0010180738],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041734576,0.00036862443,0.00017001679,0.0022177252,0.0005300367,0.00067997735,0.0007565536,0.00056560204,0.0020735797],"category_scores_gemma":[0.0634051,0.0002482753,0.00035799408,0.002041621,0.0015280202,0.0019425441,0.0009165643,0.0014162584,0.00035136606],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022199616,0.00092507224,0.9620502,0.00011869224,0.00013351353,0.00020081557,0.0016675678,0.0050373836,0.0012425921,0.0052025495,0.0010745118,0.022125226],"study_design_scores_gemma":[0.00003390042,0.00040040628,0.93552816,0.000106580934,0.00013306199,0.00095973635,0.002161467,0.050155118,0.003132174,0.0054135923,0.0019451107,0.000030731306],"about_ca_topic_score_codex":0.0035880988,"about_ca_topic_score_gemma":0.0045682425,"teacher_disagreement_score":0.0041734576,"about_ca_system_score_codex":0.00069713825,"about_ca_system_score_gemma":0.0010074242,"threshold_uncertainty_score":0.0220716},"labels":[],"label_agreement":null},{"id":"W4362584426","doi":"10.1007/s10664-022-10282-8","title":"Towards a change taxonomy for machine learning pipelines","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; York University; Ansys (Canada); Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Metadata; Replicate; Pipeline (software); Taxonomy (biology); Implementation; Data science; Fork (system call); Source code; Information retrieval; World Wide Web; Software engineering; Programming language; Ecology","score_opus":0.10899159167165356,"score_gpt":0.316721378863043,"score_spread":0.20772978719138946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4362584426","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008185324,0.0003548082,0.97752225,0.0018878196,0.00011112942,0.00058459,0.0006566226,0.0028856823,0.00781175],"genre_scores_gemma":[0.10375084,0.00045673255,0.8860087,0.0004560165,0.000096777956,0.0005365134,0.0021338311,0.0006065238,0.005954081],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9932827,0.0014674774,0.0010464598,0.0015697334,0.0020784335,0.0005551683],"domain_scores_gemma":[0.9727721,0.007659595,0.0016879733,0.0071619023,0.009246711,0.0014717394],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062137702,0.0012458237,0.0011757078,0.008686212,0.0035130822,0.00811958,0.003956741,0.004507366,0.007276356],"category_scores_gemma":[0.025822083,0.0015129816,0.003594839,0.004998873,0.0050920458,0.021204907,0.005503446,0.0073338407,0.004135376],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001382553,0.00028867414,0.012286085,0.00054337154,0.00007400111,0.00047947798,0.0029746436,0.016683828,0.0027977936,0.73124546,0.013149568,0.21933885],"study_design_scores_gemma":[0.00004303426,0.0001646883,0.0025945865,0.000341714,0.00009271681,0.00051873765,0.0011612199,0.23625317,0.0030503205,0.6670912,0.08860309,0.000085554464],"about_ca_topic_score_codex":0.011677766,"about_ca_topic_score_gemma":0.010235367,"teacher_disagreement_score":0.011677766,"about_ca_system_score_codex":0.0031418141,"about_ca_system_score_gemma":0.0054108123,"threshold_uncertainty_score":0.032861948},"labels":[],"label_agreement":null},{"id":"W4362606928","doi":"10.1007/s10664-023-10291-1","title":"Bugs in machine learning-based systems: a faultload benchmark","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Polytechnique Montréal","funders":"","keywords":"Benchmark (surveying); Computer science; Software portability; Debugging; Software bug; Software quality; Usability; Software engineering; Software; Relevance (law); Benchmarking; Machine learning; Software development; Operating system","score_opus":0.017192324944400338,"score_gpt":0.26297108381207424,"score_spread":0.2457787588676739,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4362606928","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9597038,0.0011273556,0.030170726,0.0012767354,0.00020295933,0.00008055145,0.001064521,0.0021037685,0.004269484],"genre_scores_gemma":[0.98811406,0.00012344135,0.009742257,0.00008884154,0.000039103907,0.000040093808,0.00088569435,0.00017071204,0.000795804],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99603385,0.00159428,0.0002751331,0.00054241304,0.0012448616,0.0003095715],"domain_scores_gemma":[0.92675006,0.059215926,0.0023497317,0.0072104842,0.0036368587,0.000836858],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004833115,0.00083510816,0.00054357434,0.00157792,0.00061237393,0.0006558035,0.0014568348,0.0018491122,0.002227243],"category_scores_gemma":[0.050519135,0.00032477968,0.00055213436,0.0011695019,0.0017432601,0.0019191735,0.0014054023,0.001350043,0.00032217745],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023245534,0.0023155957,0.025259994,0.0008620161,0.00029391117,0.0006188678,0.000343471,0.82974476,0.0065531465,0.023947496,0.024118425,0.08361784],"study_design_scores_gemma":[0.00028614487,0.0007395287,0.0047200355,0.000036063066,0.00004902907,0.00019094434,0.00008493661,0.97157884,0.005104201,0.015917365,0.0012740724,0.000018856801],"about_ca_topic_score_codex":0.0026405952,"about_ca_topic_score_gemma":0.0028330053,"teacher_disagreement_score":0.004833115,"about_ca_system_score_codex":0.0011536666,"about_ca_system_score_gemma":0.0010129695,"threshold_uncertainty_score":0.02556026},"labels":[],"label_agreement":null},{"id":"W4366428239","doi":"10.1007/s10664-022-10279-3","title":"Introduction to the special issue on program comprehension","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Teaching and Learning Programming","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Program comprehension; Computer science; Comprehension; Programming language; Software","score_opus":0.02009561219030038,"score_gpt":0.2880236548705743,"score_spread":0.2679280426802739,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4366428239","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00031122693,0.043194693,0.0044744913,0.07167148,0.8531728,0.00007861976,0.00077121955,0.00035909083,0.025966454],"genre_scores_gemma":[0.0011654745,0.021003205,0.0012829553,0.01781061,0.89645183,0.00011148906,0.0009974239,0.00059118716,0.060585883],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99767596,0.00045139712,0.00028436942,0.00043455628,0.0009630388,0.00019070497],"domain_scores_gemma":[0.98084325,0.008456315,0.0010050607,0.0010725827,0.005570921,0.0030518037],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037843504,0.002563614,0.0031191853,0.0072499723,0.0017286267,0.0074662208,0.0022008198,0.0045972685,0.12208732],"category_scores_gemma":[0.015227127,0.00090337923,0.0018442028,0.0038234065,0.0016332056,0.0062570604,0.004204639,0.008945497,0.056015864],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012191114,0.000029238605,0.00009338988,0.00017224664,0.000006963825,0.000020187043,0.000021671809,0.000032977656,0.00010908724,0.00094686163,0.9743871,0.024168031],"study_design_scores_gemma":[0.000010778783,0.00004481644,0.0010478521,0.0004967941,0.000016466558,0.00015702439,0.000052595184,0.00010245739,0.00008040193,0.004257913,0.9937144,0.000018421491],"about_ca_topic_score_codex":0.0010548603,"about_ca_topic_score_gemma":0.0027397012,"teacher_disagreement_score":0.12208732,"about_ca_system_score_codex":0.0015935593,"about_ca_system_score_gemma":0.0023189124,"threshold_uncertainty_score":0.40842283},"labels":[],"label_agreement":null},{"id":"W4366990909","doi":"10.1007/s10664-023-10292-0","title":"Ranking code clones to support maintenance activities","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Queen's University","funders":"","keywords":"clone (Java method); Code refactoring; Software maintenance; Commit; Computer science; Java; Code (set theory); Software evolution; Code reuse; Programming language; Software quality; Software system; Software development; Software; Database; Biology; Software construction; Genetics; Gene","score_opus":0.036875034545710464,"score_gpt":0.3080833253675743,"score_spread":0.2712082908218639,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4366990909","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97322243,0.00075643236,0.020531673,0.00022252437,0.00004641489,0.00010598002,0.0007702247,0.0022452248,0.0020990598],"genre_scores_gemma":[0.9677697,0.00014689381,0.028804518,0.00004194994,0.000024094052,0.000036482557,0.0018235731,0.0001670525,0.0011857331],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99714965,0.0005845953,0.00019691678,0.00039122422,0.0014340053,0.00024355525],"domain_scores_gemma":[0.96125376,0.019230835,0.0051490488,0.0033992068,0.009621818,0.0013454024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017119055,0.0007126986,0.00050046586,0.0058109774,0.0005086603,0.0013753604,0.0008666951,0.001088642,0.0016891026],"category_scores_gemma":[0.032727152,0.00023923366,0.0005256946,0.0024538508,0.00031791348,0.0018504066,0.0005179546,0.00053505687,0.0006609417],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001470075,0.0007238979,0.42865133,0.00048137424,0.00022574578,0.0002882001,0.00044821863,0.015972309,0.0391219,0.0019387732,0.0066592377,0.5040189],"study_design_scores_gemma":[0.00043548076,0.003972293,0.39552104,0.00023063662,0.00089737476,0.0016328163,0.0014049537,0.4984501,0.07879375,0.008656189,0.009847863,0.00015741137],"about_ca_topic_score_codex":0.004368438,"about_ca_topic_score_gemma":0.010970183,"teacher_disagreement_score":0.0058109774,"about_ca_system_score_codex":0.0007971598,"about_ca_system_score_gemma":0.001615535,"threshold_uncertainty_score":0.009053528},"labels":[],"label_agreement":null},{"id":"W4377139078","doi":"10.1007/s10664-023-10300-3","title":"Learning to Predict Code Review Completion Time In Modern Code Review","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Code review; Computer science; Context (archaeology); Coding (social sciences); Code (set theory); Process (computing); Best practice; Quality (philosophy); Software engineering; Software quality; Software development; Software","score_opus":0.03732351305167277,"score_gpt":0.3175656528412004,"score_spread":0.28024213978952767,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4377139078","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97085774,0.0005015635,0.024977919,0.0004826365,0.00007640404,0.00009121067,0.00073908555,0.00081076537,0.0014627957],"genre_scores_gemma":[0.9864433,0.000109528664,0.01052204,0.00007837543,0.00005275942,0.000058031488,0.001357423,0.000048826045,0.0013296094],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9984358,0.0005547443,0.00014111439,0.00040609972,0.00031867958,0.00014356249],"domain_scores_gemma":[0.9263188,0.05645147,0.0065001417,0.002237249,0.006292188,0.0022002396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038634534,0.00057129253,0.00044494582,0.0019060584,0.00035848533,0.0009150347,0.00061305193,0.00091356545,0.0022031676],"category_scores_gemma":[0.056065157,0.00026605165,0.0005608508,0.0008521314,0.00025194953,0.0014368958,0.00062250666,0.0016975058,0.0011637665],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015603822,0.0022702147,0.6518449,0.00023432364,0.0002664933,0.00011012325,0.00028702433,0.055657864,0.0027719259,0.0007650135,0.01012071,0.274111],"study_design_scores_gemma":[0.00010830449,0.0010663455,0.13425016,0.00006630853,0.00011230194,0.00014701484,0.00018673335,0.85565615,0.0035169055,0.0032235596,0.0016208882,0.00004542145],"about_ca_topic_score_codex":0.00588287,"about_ca_topic_score_gemma":0.010571091,"teacher_disagreement_score":0.00588287,"about_ca_system_score_codex":0.0008089977,"about_ca_system_score_gemma":0.0015613656,"threshold_uncertainty_score":0.020432115},"labels":[],"label_agreement":null},{"id":"W4377140397","doi":"10.1007/s10664-023-10313-y","title":"Relationship between diversity of collaborative group members’ race and ethnicity and the frequency of their collaborative contributions in GitHub","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; University of Waterloo","funders":"","keywords":"Ethnic group; Diversity (politics); Race (biology); Empirical research; Cultural diversity; Collaborative learning; Computer science; Knowledge management; Sociology; Gender studies; Mathematics","score_opus":0.029149707248459115,"score_gpt":0.28394597866501314,"score_spread":0.25479627141655403,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4377140397","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"incentives","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"incentives","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9983089,0.000027093129,0.000071384085,0.000069256224,0.000004408514,0.0000047343246,0.00006999852,0.0000069012435,0.0014373843],"genre_scores_gemma":[0.9989215,0.00002165013,0.00008613761,0.000019645482,0.0000053435547,0.0000068380245,0.0000699034,0.000011201004,0.000857804],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9985158,0.0006303565,0.00010296137,0.00020930712,0.0002163198,0.00032536223],"domain_scores_gemma":[0.98188055,0.008738727,0.004089034,0.0011525707,0.0015150416,0.002624111],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0017641439,0.00014001674,0.00020279195,0.0011326429,0.0011840848,0.0018537942,0.00067146117,0.00055231305,0.0065497593],"category_scores_gemma":[0.015556182,0.00014223126,0.00019297916,0.0013196837,0.0007604083,0.0010774353,0.0018465214,0.0008411518,0.00086869136],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035864694,0.00026412908,0.9706626,0.000029843173,0.000068714704,0.00018024685,0.014572981,0.00020676028,0.00077749434,0.0004966211,0.000872186,0.011509693],"study_design_scores_gemma":[0.000007616087,0.000065901586,0.9755224,0.000025478492,0.000025644282,0.00017595793,0.021632312,0.0009390707,0.00031404587,0.00032009304,0.0009580458,0.000013479953],"about_ca_topic_score_codex":0.017359007,"about_ca_topic_score_gemma":0.024409793,"teacher_disagreement_score":0.9982359,"about_ca_system_score_codex":0.00050112006,"about_ca_system_score_gemma":0.0007443513,"threshold_uncertainty_score":0.034515977},"labels":[],"label_agreement":null},{"id":"W4378072131","doi":"10.1007/s10664-023-10287-x","title":"Rubbing salt in the wound? A large-scale investigation into the effects of refactoring on security","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"Natural Sciences and Engineering Research Council of Canada; Ministero dell’Istruzione, dell’Università e della Ricerca; Università degli Studi di Salerno; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Code refactoring; Computer science; Technical debt; Source code; Software engineering; Software; Code (set theory); Source lines of code; Software development; Computer security; Programming language; Set (abstract data type)","score_opus":0.017211188368017438,"score_gpt":0.2799320263002199,"score_spread":0.26272083793220247,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378072131","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99636227,0.00029024578,0.002251615,0.00027835625,0.000008160935,0.000054812957,0.00015603905,0.000049283153,0.0005490627],"genre_scores_gemma":[0.99580646,0.00025210134,0.0032043816,0.00013501362,0.00001387388,0.00004261137,0.00024373429,0.00003711621,0.0002645887],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98595726,0.007466434,0.000960359,0.0017568826,0.0033615925,0.00049741706],"domain_scores_gemma":[0.62582564,0.29678512,0.037133045,0.021596598,0.01689694,0.0017627162],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0164219,0.0004239197,0.0004021128,0.0028080093,0.0008084961,0.0013913534,0.0010741246,0.0006583554,0.0014700937],"category_scores_gemma":[0.12210085,0.00030998778,0.00074050034,0.0029736722,0.0016843796,0.0029653993,0.0016046123,0.0013287719,0.0004252694],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058758416,0.002518467,0.8291266,0.00067656,0.00043219904,0.0006063558,0.007006134,0.0028860162,0.0049778833,0.0013299008,0.0021847792,0.14766756],"study_design_scores_gemma":[0.000060522012,0.0019230022,0.9670313,0.00038794678,0.00024499965,0.0004213318,0.0074517685,0.011387828,0.006628375,0.0012721225,0.0031261311,0.00006462959],"about_ca_topic_score_codex":0.0034487941,"about_ca_topic_score_gemma":0.0046633584,"teacher_disagreement_score":0.0164219,"about_ca_system_score_codex":0.0010560267,"about_ca_system_score_gemma":0.0012795074,"threshold_uncertainty_score":0.08684832},"labels":[],"label_agreement":null},{"id":"W4378190425","doi":"10.1007/s10664-023-10325-8","title":"18 million links in commit messages: purpose, evolution, and decay","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Wikis in Education and Collaboration","field":"Social Sciences","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Precursory Research for Embryonic Science and Technology; Japan Society for the Promotion of Science; McGill University","keywords":"Commit; Computer science; Database","score_opus":0.03138405792739127,"score_gpt":0.3356004717254505,"score_spread":0.3042164137980592,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378190425","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9931846,0.0002795775,0.0017176252,0.00037121458,0.000047032616,0.000025126437,0.001671523,0.00020435339,0.0024990188],"genre_scores_gemma":[0.9948197,0.00009804111,0.0011783584,0.00005870482,0.000036762613,0.000037124868,0.0015203707,0.00011560927,0.002135419],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99259114,0.0026838118,0.0006830161,0.0012168061,0.0022287455,0.0005964748],"domain_scores_gemma":[0.7408351,0.17542782,0.029897206,0.02135698,0.026482202,0.0060007516],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.008864168,0.00036699043,0.0004369394,0.008966397,0.0018462561,0.0035117986,0.0012354313,0.0018939107,0.0042859404],"category_scores_gemma":[0.14240319,0.00085935293,0.0005257647,0.007420431,0.0018700128,0.0050607016,0.0026663481,0.0026362238,0.0021077923],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043066184,0.0005082176,0.9229735,0.00017764718,0.00018531265,0.00012528528,0.006466943,0.0009473717,0.0020713415,0.0041776183,0.0034841204,0.058452018],"study_design_scores_gemma":[0.00004147185,0.00023540642,0.9664923,0.00016359209,0.00022689608,0.00051545975,0.0051510227,0.010034489,0.003409204,0.0060918764,0.007542857,0.00009546183],"about_ca_topic_score_codex":0.006785049,"about_ca_topic_score_gemma":0.0097062625,"teacher_disagreement_score":0.9910336,"about_ca_system_score_codex":0.0011538838,"about_ca_system_score_gemma":0.0012998367,"threshold_uncertainty_score":0.046878755},"labels":[],"label_agreement":null},{"id":"W4378474030","doi":"10.1007/s10664-023-10317-8","title":"Using the uniqueness of global identifiers to determine the provenance of Python software source code","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Identifier; Python (programming language); Computer science; Source code; Software; Open source; Source lines of code; Database; Programming language; Operating system","score_opus":0.05727007110122434,"score_gpt":0.33277357167196664,"score_spread":0.2755035005707423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378474030","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.66259,0.00045204573,0.32322943,0.00062896113,0.00017198604,0.00019985826,0.0020370928,0.0011579938,0.009532563],"genre_scores_gemma":[0.9309652,0.00015145939,0.06631658,0.00006989756,0.0000450957,0.0000984898,0.0010390227,0.0002521086,0.001062152],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98975414,0.0034869919,0.0010934989,0.0018070904,0.0032627871,0.0005953431],"domain_scores_gemma":[0.8834149,0.05284267,0.017014861,0.02336885,0.021308923,0.002049863],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013666544,0.00032536977,0.00054424565,0.0048348396,0.0017105783,0.0032360416,0.0008765295,0.0010549532,0.0017588424],"category_scores_gemma":[0.1471189,0.0005533031,0.00058287865,0.0037840733,0.0023646825,0.009539221,0.004887024,0.0021877591,0.00060321327],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083066087,0.000186664,0.6576696,0.0005456737,0.00023421776,0.0005899842,0.011130986,0.0092903925,0.016691424,0.11319984,0.0034025926,0.18622808],"study_design_scores_gemma":[0.0001973177,0.0006640716,0.28691274,0.000862926,0.0004738396,0.002018934,0.009961264,0.22092894,0.076261185,0.35676184,0.044577908,0.0003790948],"about_ca_topic_score_codex":0.0045536123,"about_ca_topic_score_gemma":0.0054886835,"teacher_disagreement_score":0.013666544,"about_ca_system_score_codex":0.0010733886,"about_ca_system_score_gemma":0.003994755,"threshold_uncertainty_score":0.07227641},"labels":[],"label_agreement":null},{"id":"W4384039146","doi":"10.1007/s10664-023-10342-7","title":"BTLink : automatic link recovery between issues and commits based on pre-trained BERT model","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Science Foundation of Jiangsu Province; National Natural Science Foundation of China","keywords":"Commit; Computer science; Traceability; Identifier; Software; Software engineering; Classifier (UML); Data mining; Machine learning; Artificial intelligence; Database; Programming language","score_opus":0.035326283152714626,"score_gpt":0.3095404327340525,"score_spread":0.2742141495813379,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384039146","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10902684,0.0015504307,0.6652384,0.0012634018,0.0012209797,0.00046096518,0.007782446,0.20786718,0.00558937],"genre_scores_gemma":[0.6647153,0.0004520453,0.277741,0.0007173288,0.0003929984,0.00042155656,0.028407821,0.0043125106,0.02283936],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986092,0.00019505218,0.00007984149,0.00047104014,0.00046276406,0.00018206576],"domain_scores_gemma":[0.9948651,0.0020262145,0.0004149918,0.0013185748,0.0010122501,0.00036288096],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017881214,0.0019540398,0.0011167949,0.00275898,0.0007584828,0.0015498401,0.0034171427,0.0022606829,0.009170246],"category_scores_gemma":[0.010211979,0.00077130797,0.0011090507,0.0014065673,0.0004560624,0.004315712,0.0021623848,0.0034783366,0.007935707],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021460983,0.0014413932,0.018981997,0.00053729466,0.00032995737,0.00064178376,0.00021431765,0.12913181,0.022082854,0.0039258837,0.13038598,0.6901808],"study_design_scores_gemma":[0.00004257039,0.00009551092,0.0011692649,0.000017534834,0.000029356976,0.00005938297,0.000029963981,0.987586,0.0043272646,0.0034476724,0.0031764824,0.000019137575],"about_ca_topic_score_codex":0.012933578,"about_ca_topic_score_gemma":0.020960782,"teacher_disagreement_score":0.012933578,"about_ca_system_score_codex":0.0008777447,"about_ca_system_score_gemma":0.0024043096,"threshold_uncertainty_score":0.030677497},"labels":[],"label_agreement":null},{"id":"W4384408554","doi":"10.1007/s10664-023-10345-4","title":"Towards a taxonomy of Roxygen documentation in R packages","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"National Research Council Canada; Natural Sciences and Engineering Research Council of Canada; University of Saskatchewan","keywords":"Documentation; Software documentation; Computer science; Taxonomy (biology); Software engineering; Card sorting; Software; World Wide Web; Reuse; Software development; Programming language; Engineering; Software development process; Task (project management); Systems engineering; Ecology","score_opus":0.16462003362110883,"score_gpt":0.3920678767072386,"score_spread":0.22744784308612975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384408554","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31374785,0.013905873,0.608149,0.00657413,0.00033944074,0.001750335,0.011784328,0.0060758684,0.03767315],"genre_scores_gemma":[0.29858887,0.0059937616,0.66804063,0.0012497904,0.00014156803,0.0016264586,0.015311494,0.0024678449,0.0065795705],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.962212,0.01757798,0.0070610675,0.0032196636,0.008717292,0.0012119306],"domain_scores_gemma":[0.8425208,0.0825417,0.023505798,0.017498095,0.03179529,0.0021383134],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02209759,0.0009738485,0.0009780308,0.026838448,0.0028042372,0.007157727,0.002133514,0.0020029233,0.0030141913],"category_scores_gemma":[0.08303028,0.0011131916,0.0012792776,0.022056518,0.003607173,0.01049127,0.004808155,0.0022550935,0.0019381617],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029780885,0.0002971889,0.1344787,0.0071389447,0.00017284445,0.0023083782,0.10167928,0.0035410728,0.013695425,0.20363738,0.035953127,0.49679995],"study_design_scores_gemma":[0.00005051386,0.00030941647,0.076068416,0.008975098,0.00018211987,0.00681029,0.041528318,0.017947836,0.0070256707,0.08198571,0.75870526,0.00041130092],"about_ca_topic_score_codex":0.0046113753,"about_ca_topic_score_gemma":0.0049202573,"teacher_disagreement_score":0.9779024,"about_ca_system_score_codex":0.0031341459,"about_ca_system_score_gemma":0.006293536,"threshold_uncertainty_score":0.11686462},"labels":[],"label_agreement":null},{"id":"W4385740122","doi":"10.1007/s10664-023-10309-8","title":"What have we learned? A conceptual framework on New Zealand software professionals and companies’ response to COVID-19","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Supply Chain Resilience and Risk Management","field":"Business, Management and Accounting","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Workaround; Thematic analysis; Software; Software development; Work (physics); Personal software process; Software peer review; Exploratory research; Software review; Knowledge management; Engineering; Business; Computer science; Qualitative research; Software construction; Sociology","score_opus":0.051342450730401896,"score_gpt":0.33055182627901813,"score_spread":0.27920937554861625,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385740122","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37728348,0.0069567314,0.014180339,0.3790492,0.00073262246,0.00057273614,0.00035730918,0.00006913463,0.22079845],"genre_scores_gemma":[0.9771782,0.0024369522,0.004236703,0.005118204,0.00008960422,0.00025226315,0.00008563918,0.00002788066,0.010574594],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9898301,0.005079486,0.00046513532,0.00081474194,0.001860788,0.0019497005],"domain_scores_gemma":[0.95797807,0.019164411,0.0045017605,0.0013978444,0.008906419,0.008051396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024265258,0.00063768,0.00052775675,0.0056077866,0.009169536,0.024111727,0.0033018384,0.0048513752,0.008364028],"category_scores_gemma":[0.04431456,0.0004551048,0.00050705747,0.005677828,0.045248356,0.026958315,0.008648404,0.006693044,0.0006151452],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005953487,0.00016872145,0.017280936,0.00033939857,0.000022072547,0.000575198,0.5965036,0.0005208187,0.00020942317,0.33966705,0.0073242895,0.037328895],"study_design_scores_gemma":[0.000025505458,0.00008427907,0.016904118,0.0014797983,0.000018384826,0.00017732366,0.79687065,0.0009769194,0.00012981023,0.09866918,0.08458826,0.00007567288],"about_ca_topic_score_codex":0.31446818,"about_ca_topic_score_gemma":0.26495388,"teacher_disagreement_score":0.31446818,"about_ca_system_score_codex":0.03789409,"about_ca_system_score_gemma":0.07430318,"threshold_uncertainty_score":0.6252755},"labels":[],"label_agreement":null},{"id":"W4385952926","doi":"10.1007/s10664-023-10323-w","title":"XSnare: application-specific client-side cross-site scripting protection","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Web Application Security Vulnerabilities","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Cross-site scripting; Exploit; Computer science; Scripting language; Client-side; Context (archaeology); Computer security; Overhead (engineering); World Wide Web; Database; Web page; Web application security; Operating system; Web development","score_opus":0.03303138290703469,"score_gpt":0.2844141654592613,"score_spread":0.2513827825522266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385952926","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.56492454,0.0006072441,0.23805846,0.0016316369,0.00015700443,0.0008798676,0.008321214,0.09626298,0.08915708],"genre_scores_gemma":[0.9224164,0.0002321678,0.053204924,0.0003599206,0.000038006332,0.00024222994,0.0058793435,0.0028602416,0.014766692],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99636275,0.0011347798,0.00023424807,0.0004016808,0.0015831322,0.00028337236],"domain_scores_gemma":[0.98393136,0.0061046937,0.0017958408,0.006510382,0.0012419873,0.0004156688],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049358117,0.00046566647,0.0002688153,0.0014927335,0.0005630334,0.001320245,0.001257625,0.00089270703,0.008769098],"category_scores_gemma":[0.02040896,0.00036727745,0.0003605722,0.0009863535,0.0011289846,0.0021490147,0.0019000075,0.0015158245,0.0024334649],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013833678,0.0025805333,0.11114945,0.00093952275,0.00026317057,0.0006822813,0.0030033921,0.015167483,0.036731634,0.07386879,0.06547126,0.68875915],"study_design_scores_gemma":[0.0004915804,0.0025344037,0.27955547,0.00071920425,0.00045768885,0.005073721,0.0015203544,0.18023989,0.21048991,0.08197499,0.23660004,0.0003428052],"about_ca_topic_score_codex":0.0019889427,"about_ca_topic_score_gemma":0.0017329422,"teacher_disagreement_score":0.008769098,"about_ca_system_score_codex":0.0006572198,"about_ca_system_score_gemma":0.0016036257,"threshold_uncertainty_score":0.029335558},"labels":[],"label_agreement":null},{"id":"W4386140507","doi":"10.1007/s10664-023-10363-2","title":"A comparison of reinforcement learning frameworks for software testing tasks","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Regression testing; Software performance testing; Software engineering; System integration testing; Software reliability testing; Software development; Adaptation (eye); Context (archaeology); Test strategy; Reinforcement learning; Software inspection; Manual testing; Software testing; Software construction; Machine learning; Software; Software quality; Programming language","score_opus":0.06973939181903052,"score_gpt":0.35039542706618215,"score_spread":0.28065603524715166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386140507","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2330351,0.003478516,0.7472488,0.00091836974,0.00019646679,0.00047264292,0.00011730628,0.0032867563,0.011246057],"genre_scores_gemma":[0.86274034,0.0008000259,0.13340698,0.00012465256,0.00004897991,0.00025318997,0.00013815227,0.0002589108,0.0022287855],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99500656,0.0027374083,0.0002599138,0.00036493954,0.0012656392,0.00036561972],"domain_scores_gemma":[0.9532519,0.03867261,0.001200451,0.002481481,0.0031976039,0.001195987],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010138228,0.0008326941,0.0012519606,0.0014592473,0.00051089644,0.0015521116,0.0026852398,0.0016819366,0.0024643452],"category_scores_gemma":[0.038140696,0.0004474191,0.00072570733,0.00084340427,0.0010609177,0.0025792255,0.001626477,0.0020853956,0.00049231516],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032479183,0.0019538237,0.0064355596,0.0005297229,0.00020282035,0.000062329906,0.0005686759,0.40932897,0.0026341113,0.042751335,0.0021248013,0.5301599],"study_design_scores_gemma":[0.00020729414,0.0005641889,0.0022070776,0.00005893735,0.000051357805,0.00003704463,0.000079376594,0.9826974,0.0010144345,0.011843474,0.0012069936,0.00003234466],"about_ca_topic_score_codex":0.007389067,"about_ca_topic_score_gemma":0.005608095,"teacher_disagreement_score":0.010138228,"about_ca_system_score_codex":0.002282094,"about_ca_system_score_gemma":0.0028369993,"threshold_uncertainty_score":0.053616703},"labels":[],"label_agreement":null},{"id":"W4386245573","doi":"10.1007/s10664-023-10348-1","title":"On practitioners’ concerns when adopting service mesh frameworks","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Service (business); Documentation; Data science; Microservices; Service provider; Domain (mathematical analysis); Computer security; Cloud computing; Business","score_opus":0.028533688769403535,"score_gpt":0.2867786734879754,"score_spread":0.2582449847185719,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386245573","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46036384,0.0028227754,0.04804262,0.40070254,0.001247223,0.00035135887,0.00007125189,0.00033270495,0.08606569],"genre_scores_gemma":[0.9656141,0.00058853696,0.011061204,0.020124426,0.00018806322,0.00013937899,0.000026297577,0.00011112579,0.0021467768],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.7612909,0.13686319,0.010613385,0.006632672,0.07309573,0.011504184],"domain_scores_gemma":[0.3269613,0.5043423,0.029910257,0.016801935,0.11049724,0.011486923],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.22726905,0.00040788137,0.00048493964,0.003338796,0.007638996,0.013210956,0.0034251558,0.010309779,0.0046882527],"category_scores_gemma":[0.55887765,0.00095042743,0.0006954647,0.0028851759,0.010083873,0.013848767,0.005859694,0.014290919,0.000718838],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046195317,0.00077227334,0.10868945,0.0010904403,0.0001788832,0.0034891842,0.39458993,0.0032360193,0.0045884084,0.1785694,0.044864528,0.2594695],"study_design_scores_gemma":[0.00017630617,0.00096170715,0.04883016,0.0050094603,0.00023306563,0.0029729544,0.5939792,0.009635353,0.0046127373,0.1182638,0.21498357,0.00034173115],"about_ca_topic_score_codex":0.01730445,"about_ca_topic_score_gemma":0.025658986,"teacher_disagreement_score":0.22726905,"about_ca_system_score_codex":0.01577252,"about_ca_system_score_gemma":0.02820492,"threshold_uncertainty_score":0.95291483},"labels":[],"label_agreement":null},{"id":"W4386715810","doi":"10.1007/s10664-023-10373-0","title":"Investigating developers’ perception on software testability and its effects","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; Dalhousie University","funders":"","keywords":"Testability; Code smell; Software engineering; Computer science; Empirical research; Software quality; Software development; Software; Engineering; Reliability engineering; Programming language","score_opus":0.03623200916931937,"score_gpt":0.2923139179714733,"score_spread":0.256081908802154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386715810","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9978999,0.000047471774,0.0003645109,0.00008156583,0.0000035414564,0.0000052584687,0.000016245893,0.0000074017175,0.0015740468],"genre_scores_gemma":[0.99954396,0.000023436918,0.00019042355,0.00001556655,0.000002894033,0.0000034689535,0.000021671678,0.0000037363955,0.00019487648],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99600947,0.0018055593,0.0003344837,0.00030078168,0.0012772455,0.00027245443],"domain_scores_gemma":[0.7585537,0.19514947,0.023039693,0.0048852884,0.013427385,0.0049444474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006018847,0.00022811462,0.0001781391,0.00089085894,0.00026088394,0.0011232396,0.00031641367,0.0005974649,0.002964831],"category_scores_gemma":[0.10023458,0.00019620733,0.00028285215,0.0005612292,0.0005712501,0.0011524975,0.00077338493,0.00078457623,0.00021723019],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071028597,0.00069474557,0.93723565,0.0001624863,0.00009467489,0.00030570818,0.018544534,0.00049754104,0.00915129,0.00048212276,0.00033416055,0.031786766],"study_design_scores_gemma":[0.000025039522,0.0005424721,0.9885076,0.000038365473,0.00005542793,0.00015934216,0.0066933045,0.001353645,0.0016491835,0.0002883528,0.00066646124,0.00002077711],"about_ca_topic_score_codex":0.0037080683,"about_ca_topic_score_gemma":0.0047599524,"teacher_disagreement_score":0.006018847,"about_ca_system_score_codex":0.00058875495,"about_ca_system_score_gemma":0.00064226484,"threshold_uncertainty_score":0.031831086},"labels":[],"label_agreement":null},{"id":"W4386781823","doi":"10.1007/s10664-023-10347-2","title":"A study of documentation for software architecture","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Documentation; Software documentation; Computer science; Internal documentation; Context (archaeology); World Wide Web; Software engineering; Software; Software system; Information retrieval; Programming language; Software construction","score_opus":0.0365751899518581,"score_gpt":0.329109199373997,"score_spread":0.2925340094221389,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386781823","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94915724,0.0011259953,0.009754143,0.0018739762,0.00002414026,0.000036760597,0.000055366057,0.00007344155,0.037898943],"genre_scores_gemma":[0.99092126,0.00047830862,0.0038857593,0.00008833377,0.000013489673,0.000016901811,0.000058839832,0.00003309564,0.004504045],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.99741864,0.0016070567,0.000117437,0.00011325624,0.00058014435,0.00016342825],"domain_scores_gemma":[0.92338234,0.06072633,0.00497383,0.0047759106,0.0049689324,0.0011726525],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0035713625,0.00023075256,0.00022123342,0.002232267,0.0024750682,0.0023407694,0.00073453336,0.0010417814,0.0033631779],"category_scores_gemma":[0.060008924,0.00033284153,0.00027852497,0.003920404,0.002632151,0.0049281195,0.001157751,0.0018654552,0.00032295243],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002987343,0.0015650856,0.20147085,0.0005599684,0.000051397456,0.001681837,0.08727778,0.0061358586,0.0040943916,0.43517914,0.007519256,0.25416574],"study_design_scores_gemma":[0.00022332859,0.001811058,0.34899426,0.0018650026,0.00018178973,0.005037546,0.10879419,0.056143153,0.010119697,0.30524445,0.161424,0.00016153062],"about_ca_topic_score_codex":0.008598512,"about_ca_topic_score_gemma":0.011250468,"teacher_disagreement_score":0.9964286,"about_ca_system_score_codex":0.0024824527,"about_ca_system_score_gemma":0.0039718268,"threshold_uncertainty_score":0.0188874},"labels":[],"label_agreement":null},{"id":"W4386982649","doi":"10.1007/s10664-023-10380-1","title":"Is GitHub’s Copilot as bad as humans at introducing vulnerabilities in code?","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":101,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Vulnerability (computing); Code (set theory); Process (computing); Computer security; Perspective (graphical); Secure coding; Software; Software engineering; Software security assurance; Artificial intelligence; Operating system; Information security; Programming language","score_opus":0.03474644801521013,"score_gpt":0.3154596742478399,"score_spread":0.28071322623262973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386982649","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6374408,0.0043500513,0.044341397,0.17143948,0.0025930752,0.00015216068,0.0013690192,0.010164866,0.12814924],"genre_scores_gemma":[0.95042783,0.0011409206,0.018258484,0.01746825,0.000322523,0.00005797809,0.00081321405,0.0023744698,0.009136404],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9867927,0.0056361733,0.00037819182,0.0013278206,0.004550868,0.0013143821],"domain_scores_gemma":[0.9246243,0.035595886,0.008131279,0.016111314,0.011203392,0.004333802],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012840184,0.0007556767,0.00058057706,0.0021953646,0.0019537748,0.004201843,0.00148113,0.0035679808,0.0075339377],"category_scores_gemma":[0.11352643,0.0005736612,0.00046579458,0.0016228828,0.005434384,0.00982682,0.0030052043,0.0032896413,0.0031157413],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018738054,0.0006240356,0.2007792,0.0009830421,0.0005266014,0.0011321205,0.0146606015,0.0045946348,0.007666979,0.07533218,0.2782953,0.41353157],"study_design_scores_gemma":[0.00045318308,0.0016833603,0.19890307,0.0025186313,0.00073039427,0.0057502743,0.035737615,0.031470668,0.025863752,0.21326347,0.48300874,0.0006168806],"about_ca_topic_score_codex":0.012724229,"about_ca_topic_score_gemma":0.018767169,"teacher_disagreement_score":0.012840184,"about_ca_system_score_codex":0.001564931,"about_ca_system_score_gemma":0.0043486687,"threshold_uncertainty_score":0.0679062},"labels":[],"label_agreement":null},{"id":"W4387435361","doi":"10.1007/s10664-023-10364-1","title":"On the effectiveness of log representation for log-based anomaly detection","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Representation (politics); Computer science; Workflow; Web log analysis software; Context (archaeology); Data mining; Anomaly detection; Feature (linguistics); External Data Representation; Heuristic; Log-log plot; Software; Binary logarithm; Artificial intelligence; Database; Mathematics; World Wide Web; Programming language; Web service","score_opus":0.025893510235824126,"score_gpt":0.2845307926426779,"score_spread":0.25863728240685374,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387435361","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21865134,0.0020301284,0.7701521,0.0012806862,0.00016015061,0.000094338946,0.00038873995,0.0026182707,0.004624315],"genre_scores_gemma":[0.91640836,0.0005272982,0.08118993,0.00014789958,0.00017967263,0.000037912097,0.00035462755,0.00010937782,0.0010447975],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9946807,0.002989281,0.0002789177,0.00055620907,0.0011952653,0.00029966337],"domain_scores_gemma":[0.89278704,0.097281195,0.0019427021,0.005023359,0.002555506,0.00041022396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008194394,0.0009080081,0.001137059,0.0024953494,0.0005356799,0.0024014544,0.0014241658,0.0017055682,0.0015600823],"category_scores_gemma":[0.07160185,0.00034310663,0.0005469165,0.0014428317,0.0010433662,0.004887678,0.0013762331,0.0016925129,0.0004805645],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030862098,0.00088346755,0.03161988,0.00028820045,0.00018847673,0.0002476115,0.0002879013,0.23046237,0.013751779,0.03050075,0.003991057,0.68469226],"study_design_scores_gemma":[0.00002600115,0.00015700691,0.0026493717,0.00002087212,0.000032947057,0.00016190867,0.000059604186,0.9852342,0.002954634,0.008305151,0.00037897553,0.0000194307],"about_ca_topic_score_codex":0.0038191138,"about_ca_topic_score_gemma":0.0016527803,"teacher_disagreement_score":0.008194394,"about_ca_system_score_codex":0.00062225986,"about_ca_system_score_gemma":0.00083785644,"threshold_uncertainty_score":0.04333663},"labels":[],"label_agreement":null},{"id":"W4387739952","doi":"10.1007/s10664-023-10382-z","title":"Studying the characteristics of AIOps projects on GitHub","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Baseline (sea); Sample (material); Context (archaeology); Data science; Set (abstract data type); Quality (philosophy); Software engineering; Software; Open source; Anomaly detection; Data mining","score_opus":0.05649726322564533,"score_gpt":0.29934722376021994,"score_spread":0.2428499605345746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387739952","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9965515,0.00004424418,0.00018103595,0.00008670367,0.0000032528624,0.000021583692,0.00015026152,0.000030631647,0.0029308072],"genre_scores_gemma":[0.99712783,0.00007409049,0.00041810574,0.00002496528,0.0000066614566,0.000031844924,0.0004992452,0.000051341045,0.0017659497],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9971257,0.0007610168,0.00017625783,0.0003196046,0.001052457,0.00056500756],"domain_scores_gemma":[0.9633774,0.013354419,0.009690589,0.0016434967,0.0059043425,0.0060296613],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0023103412,0.00028106396,0.00020199611,0.0048844414,0.0010439756,0.0026170711,0.0007704103,0.00045689804,0.0032711786],"category_scores_gemma":[0.027060652,0.00022999039,0.00017243077,0.0073055234,0.0007731764,0.0020954462,0.001983269,0.00084047124,0.00086264435],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020136836,0.00036891858,0.9400247,0.000097630626,0.000044611214,0.0004916952,0.00970639,0.0008172618,0.002291508,0.0014023887,0.0022119556,0.042341597],"study_design_scores_gemma":[0.0000074770915,0.00013666271,0.9820157,0.000027864822,0.000009028285,0.00019085275,0.011849159,0.0018830816,0.00049666816,0.0002876472,0.0030780565,0.000017895672],"about_ca_topic_score_codex":0.016545564,"about_ca_topic_score_gemma":0.0351885,"teacher_disagreement_score":0.9951156,"about_ca_system_score_codex":0.0017122837,"about_ca_system_score_gemma":0.002061564,"threshold_uncertainty_score":0.032898545},"labels":[],"label_agreement":null},{"id":"W4388568365","doi":"10.1007/s10664-023-10403-x","title":"On the coordination of vulnerability fixes","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Network Security and Intrusion Detection","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Vulnerability (computing); Computer security; Secure coding; Computer science; Software; Order (exchange); Internet privacy; Business; Software security assurance; Information security","score_opus":0.02498656131703465,"score_gpt":0.2621775284910033,"score_spread":0.23719096717396865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388568365","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6390635,0.0020579426,0.26946712,0.0085291015,0.0001836257,0.00025785773,0.00069316145,0.00064320245,0.07910452],"genre_scores_gemma":[0.98578656,0.00034759362,0.011166439,0.00009287723,0.000056175428,0.00006036149,0.00016813596,0.00008700099,0.0022348848],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.98869276,0.006105736,0.00069956493,0.0017962022,0.0015295036,0.0011763739],"domain_scores_gemma":[0.77579975,0.16987287,0.023495238,0.020942576,0.0064330585,0.003456436],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014784441,0.00072815135,0.0014237128,0.0036226425,0.0017067402,0.0049625416,0.0025451728,0.0023623842,0.01776405],"category_scores_gemma":[0.1692765,0.0009613631,0.0010412917,0.0030426811,0.003861707,0.008615002,0.0042728516,0.0028899712,0.00097051536],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067455857,0.00036658326,0.06374018,0.00018820558,0.00036792996,0.0004146749,0.0015256209,0.12917307,0.0015313289,0.71436435,0.0066520693,0.08100138],"study_design_scores_gemma":[0.00020432094,0.00033736628,0.033447877,0.00016474583,0.0002024857,0.00030244087,0.0018428746,0.21301696,0.0015064798,0.7416647,0.007234276,0.00007547492],"about_ca_topic_score_codex":0.009501398,"about_ca_topic_score_gemma":0.0061175544,"teacher_disagreement_score":0.01776405,"about_ca_system_score_codex":0.0031088116,"about_ca_system_score_gemma":0.003503049,"threshold_uncertainty_score":0.07818854},"labels":[],"label_agreement":null},{"id":"W4388568366","doi":"10.1007/s10664-023-10394-9","title":"An empirical comparison of ethnic and gender diversity of DevOps and non-DevOps contributions to open-source projects","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"DevOps; Diversity (politics); Ethnic group; Empirical research; Open source software; Cultural diversity; Engineering; Computer science; Political science; Software; Software deployment; Software engineering; Statistics","score_opus":0.08380178275201619,"score_gpt":0.3824720258389828,"score_spread":0.29867024308696666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388568366","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99888223,0.000025715186,0.00007779865,0.000031885902,0.000002493142,0.0000045420875,0.00004343655,5.8256376e-7,0.00093126786],"genre_scores_gemma":[0.9993414,0.000031743417,0.00004221199,0.000019529898,0.0000036449705,0.000009107157,0.00006378058,0.0000017682394,0.00048659477],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99791664,0.00081571203,0.00014014376,0.000284255,0.0004239889,0.00041926035],"domain_scores_gemma":[0.9809413,0.00751986,0.005361997,0.0007181638,0.0025311618,0.002927484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025334612,0.00018204433,0.0001669217,0.0021959434,0.0011628972,0.0014082775,0.000509157,0.0003621137,0.0043566967],"category_scores_gemma":[0.01700717,0.00016975972,0.00018300397,0.0021294097,0.0011907212,0.0015134196,0.0020150268,0.0006596558,0.00062074256],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000106154665,0.00018697757,0.970549,0.000016883312,0.00001836354,0.00009106035,0.02248895,0.00002403647,0.0005815942,0.00045096333,0.00021731002,0.005268702],"study_design_scores_gemma":[0.0000047694934,0.00008275083,0.9402339,0.000025856896,0.000008376448,0.0000964383,0.058242887,0.00010441338,0.00021754853,0.00019651493,0.00077962835,0.0000069121547],"about_ca_topic_score_codex":0.0053273267,"about_ca_topic_score_gemma":0.00956645,"teacher_disagreement_score":0.0053273267,"about_ca_system_score_codex":0.00047037963,"about_ca_system_score_gemma":0.0007313309,"threshold_uncertainty_score":0.014574528},"labels":[],"label_agreement":null},{"id":"W4389141459","doi":"10.1007/s10664-023-10389-6","title":"Silent bugs in deep learning frameworks: an empirical study of Keras and TensorFlow","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de recherche du Québec; Consortium de Recherche et d’innovation en Aérospatiale au Québec; Canadian Institute for Advanced Research","keywords":"Software bug; Computer science; Relevance (law); Debugging; Artificial intelligence; Empirical research; Deep learning; Machine learning; Software; Programming language","score_opus":0.02542377729338102,"score_gpt":0.3172849195316602,"score_spread":0.2918611422382792,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389141459","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98428345,0.0008296527,0.011164901,0.0008243346,0.000039135208,0.00004250032,0.00018922972,0.0006543831,0.001972387],"genre_scores_gemma":[0.99583143,0.00008232419,0.003395723,0.00006705879,0.000012638521,0.000015785034,0.00017944306,0.000113328846,0.00030230856],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98871917,0.0046696966,0.00068742054,0.0014012884,0.0036139083,0.0009084697],"domain_scores_gemma":[0.68801606,0.2473794,0.024937013,0.023321927,0.012175107,0.004170527],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0170036,0.0006607831,0.0005869774,0.0017123846,0.0012889195,0.0018038333,0.002579601,0.0023034262,0.0020983776],"category_scores_gemma":[0.22011997,0.0006145296,0.00061695627,0.0019721857,0.0033000768,0.008320617,0.0021248509,0.00485606,0.00033032589],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0045130397,0.0067428173,0.53478426,0.0011737962,0.0006179828,0.0010561844,0.008186323,0.08544864,0.0048960587,0.08509899,0.019933395,0.2475485],"study_design_scores_gemma":[0.0005670531,0.002477652,0.121728145,0.00049396296,0.00043844318,0.0012563339,0.0042569195,0.7412241,0.0057173823,0.1147117,0.006912648,0.00021562219],"about_ca_topic_score_codex":0.007722823,"about_ca_topic_score_gemma":0.0083760675,"teacher_disagreement_score":0.9829964,"about_ca_system_score_codex":0.001782899,"about_ca_system_score_gemma":0.0025358687,"threshold_uncertainty_score":0.08992469},"labels":[],"label_agreement":null},{"id":"W4389141545","doi":"10.1007/s10664-023-10399-4","title":"Unreproducible builds: time to fix, causes, and correlation with external ecosystem factors","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Software; Process (computing); Reproducibility; Cyclomatic complexity; Software engineering; Data science; Operating system; Statistics; Mathematics","score_opus":0.019738238128314236,"score_gpt":0.25718784017076846,"score_spread":0.23744960204245422,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389141545","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9977914,0.00025564607,0.0007366951,0.00009732312,0.0000056098916,0.00001203276,0.00018873617,0.000022112477,0.00089042063],"genre_scores_gemma":[0.99938715,0.000054721895,0.0001992388,0.0000068903655,0.0000049596983,0.0000063006833,0.00010966199,0.000011952993,0.00021915077],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99698776,0.0009901237,0.00044777605,0.00043034676,0.000716276,0.00042762636],"domain_scores_gemma":[0.7642281,0.1638295,0.04883907,0.011619904,0.00571416,0.0057692127],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0065181684,0.00029339464,0.00034346743,0.002086558,0.00054855633,0.0019713982,0.0010098846,0.0011389848,0.00473625],"category_scores_gemma":[0.09834314,0.00048948114,0.0006231088,0.0016338846,0.0012772689,0.0021106217,0.0014358924,0.0021982014,0.0005216803],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017369473,0.00018653694,0.9921523,0.00002150272,0.00008233988,0.0001050491,0.00030061373,0.0016078675,0.00020199773,0.00048565824,0.00011022834,0.0045720963],"study_design_scores_gemma":[0.00001875145,0.0001607855,0.9894917,0.000030941548,0.000097782366,0.0004359326,0.00084907736,0.00553341,0.0005151288,0.0023240703,0.00051719666,0.000025243451],"about_ca_topic_score_codex":0.0053844987,"about_ca_topic_score_gemma":0.008559096,"teacher_disagreement_score":0.9934818,"about_ca_system_score_codex":0.00088695396,"about_ca_system_score_gemma":0.0016363972,"threshold_uncertainty_score":0.03447181},"labels":[],"label_agreement":null},{"id":"W4389195248","doi":"10.1007/s10664-023-10398-5","title":"A fly in the ointment: an empirical study on the characteristics of Ethereum smart contract code weaknesses","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Rannís","keywords":"Computer science; Strengths and weaknesses; Smart contract; Generality; Categorization; Code (set theory); Source code; Coding (social sciences); Data science; Artificial intelligence; Data mining; Database; Programming language; Database transaction; Mathematics","score_opus":0.03717403338258266,"score_gpt":0.30452667689705554,"score_spread":0.26735264351447285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389195248","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99717253,0.00003973849,0.00060943974,0.00014072392,0.0000014486801,0.000016453445,0.00005732459,0.000005600246,0.001956751],"genre_scores_gemma":[0.9991014,0.000020054129,0.00024243513,0.000016477115,0.0000015225309,0.000005771283,0.00006998164,0.0000072232465,0.00053508824],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9969537,0.00075518934,0.0002844502,0.00032700197,0.0013546632,0.00032503574],"domain_scores_gemma":[0.80717593,0.121226296,0.052201323,0.005423889,0.0107344575,0.00323821],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.006660623,0.00021483631,0.00025017842,0.0032103641,0.0010328842,0.0018629982,0.00095121615,0.0013092584,0.00457973],"category_scores_gemma":[0.09434745,0.0002643815,0.00017981099,0.0031805902,0.002543729,0.0062360475,0.0015636524,0.0019164172,0.0004021075],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026541276,0.00027168222,0.96292865,0.000068621186,0.00004099559,0.00049991027,0.0045071053,0.0019091277,0.0011583768,0.01143092,0.0005880016,0.016331129],"study_design_scores_gemma":[0.000039487943,0.00048513574,0.9066699,0.00023069706,0.00008146346,0.0017335215,0.029202152,0.04012461,0.0035011615,0.012117157,0.00573767,0.00007705219],"about_ca_topic_score_codex":0.005631109,"about_ca_topic_score_gemma":0.0081153875,"teacher_disagreement_score":0.99333936,"about_ca_system_score_codex":0.001232354,"about_ca_system_score_gemma":0.0017057064,"threshold_uncertainty_score":0.035225153},"labels":[],"label_agreement":null},{"id":"W4389337865","doi":"10.1007/s10664-023-10400-0","title":"Bug characterization in machine learning-based systems","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Polytechnique Montréal","funders":"","keywords":"Computer science; Software bug; Software; Component (thermodynamics); Software system; Task (project management); Process (computing); Software engineering; Software development; Focus (optics); Software maintenance; Corrective maintenance; Machine learning; Reliability engineering; Operating system; Systems engineering; Engineering; Preventive maintenance","score_opus":0.024405773824118494,"score_gpt":0.2681081263815844,"score_spread":0.2437023525574659,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389337865","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8627934,0.0010514422,0.13099271,0.00042917317,0.000046078872,0.0001227786,0.0007064349,0.001808361,0.0020496398],"genre_scores_gemma":[0.9758084,0.00008541236,0.023015628,0.000029958948,0.000014797892,0.000027836004,0.0005729179,0.000088673114,0.00035644614],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9933207,0.0019310777,0.00095756777,0.0011000914,0.0022693456,0.0004211822],"domain_scores_gemma":[0.9147225,0.05359252,0.013566099,0.008267517,0.008839646,0.0010117553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004665167,0.0005314131,0.0005260096,0.0062786965,0.0005330492,0.0014760083,0.000950951,0.0010079887,0.0011892029],"category_scores_gemma":[0.06844034,0.00034198535,0.00055093266,0.0024883032,0.00094554835,0.0031122302,0.0009171902,0.0010315606,0.00024262554],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058314326,0.00058681634,0.58652693,0.00056847994,0.00021044513,0.00041592307,0.0011267341,0.06986167,0.008586907,0.010415968,0.0030204037,0.31809658],"study_design_scores_gemma":[0.000065717686,0.0004946084,0.15054895,0.00021645357,0.0001565505,0.0012336123,0.00058202824,0.80168235,0.011810432,0.031093016,0.0020489672,0.00006732902],"about_ca_topic_score_codex":0.003376165,"about_ca_topic_score_gemma":0.0043997243,"teacher_disagreement_score":0.0062786965,"about_ca_system_score_codex":0.0008380294,"about_ca_system_score_gemma":0.0012247891,"threshold_uncertainty_score":0.024672031},"labels":[],"label_agreement":null},{"id":"W4389685461","doi":"10.1007/s10664-023-10409-5","title":"Detection and evaluation of bias-inducing features in machine learning","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Consortium de Recherche et d’innovation en Aérospatiale au Québec","keywords":"Computer science; Machine learning; Identification (biology); Context (archaeology); Artificial intelligence; Harm; Feature (linguistics); Outcome (game theory); Data mining; Psychology","score_opus":0.06541519331576667,"score_gpt":0.314828704639924,"score_spread":0.24941351132415734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389685461","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7207711,0.0021165342,0.2713148,0.00047316443,0.000116805146,0.00016650322,0.00060262054,0.0028481663,0.00159024],"genre_scores_gemma":[0.9522358,0.00016833728,0.046241336,0.0000591415,0.00004062802,0.000041118576,0.0007300905,0.00014978569,0.00033377288],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99698097,0.0010600631,0.00024943912,0.00046705228,0.001013933,0.00022855138],"domain_scores_gemma":[0.95513123,0.032293078,0.0033840367,0.0035122402,0.004894659,0.00078477414],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0070135454,0.0006388181,0.0008907859,0.002285584,0.0005458346,0.0018182165,0.00093910063,0.001520806,0.0007912155],"category_scores_gemma":[0.045618672,0.00027922212,0.0005022235,0.0010861736,0.0006648891,0.0014541751,0.0013233848,0.0012220296,0.0003292823],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026798614,0.00086045265,0.1793462,0.00073605473,0.00043287282,0.00031968398,0.00027886097,0.065958604,0.04713005,0.009052393,0.0045442693,0.6886607],"study_design_scores_gemma":[0.00016449625,0.00047093586,0.02734446,0.00009220833,0.00018215821,0.0002572579,0.000098620265,0.9121817,0.04679354,0.011150645,0.0012257425,0.00003830316],"about_ca_topic_score_codex":0.0013227339,"about_ca_topic_score_gemma":0.0017932251,"teacher_disagreement_score":0.99298644,"about_ca_system_score_codex":0.00074786047,"about_ca_system_score_gemma":0.0012014123,"threshold_uncertainty_score":0.037091613},"labels":[],"label_agreement":null},{"id":"W4389804693","doi":"10.1007/s10664-023-10376-x","title":"On the intuitive comprehensibility of contribution links in goal models: an experimental study","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Cognitive and psychological constructs research","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Psychology","score_opus":0.138767554850462,"score_gpt":0.42106434379206314,"score_spread":0.28229678894160115,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389804693","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99346465,0.000043117,0.0038855607,0.00008043386,0.000014261037,0.00010395128,0.000033704968,0.000033696422,0.0023406274],"genre_scores_gemma":[0.9953151,0.00005440011,0.0036223999,0.00005553901,0.0000126986415,0.00012030385,0.000076726625,0.000053182062,0.0006896238],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9950068,0.0028477393,0.00046950765,0.00076556404,0.0007592439,0.00015109431],"domain_scores_gemma":[0.599026,0.37463665,0.009183952,0.012160628,0.0038211145,0.0011716346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010016327,0.0008091678,0.00044402172,0.0009735846,0.0004829733,0.003326253,0.001293033,0.0016693304,0.0088137295],"category_scores_gemma":[0.21381864,0.00078807364,0.00049723196,0.00047029418,0.0016971122,0.005502438,0.0019566398,0.002580585,0.0005669697],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.018742032,0.03324264,0.21930028,0.0033093453,0.0008498855,0.002061722,0.17586234,0.035453256,0.22932738,0.03976624,0.0041266833,0.23795816],"study_design_scores_gemma":[0.0034441906,0.024645431,0.369556,0.0009756175,0.001571186,0.002814568,0.04410236,0.39668527,0.080487266,0.06569054,0.009349062,0.0006786076],"about_ca_topic_score_codex":0.00076862506,"about_ca_topic_score_gemma":0.00043246185,"teacher_disagreement_score":0.010016327,"about_ca_system_score_codex":0.00047925313,"about_ca_system_score_gemma":0.00046267017,"threshold_uncertainty_score":0.05297202},"labels":[],"label_agreement":null},{"id":"W4390144898","doi":"10.1007/s10664-023-10431-7","title":"A multi-objective effort-aware approach for early code review prediction and prioritization","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Code review; Code (set theory); Machine learning; Sorting; Prioritization; Empirical research; Source code; Task (project management); Artificial intelligence; Software; Static program analysis; Software engineering; Software development; Management science; Algorithm; Engineering; Systems engineering; Programming language","score_opus":0.039604791623415604,"score_gpt":0.3068962005751302,"score_spread":0.26729140895171455,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390144898","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15232903,0.0019231315,0.8350382,0.0010687253,0.00018256412,0.0004923899,0.001295985,0.0038696202,0.0038004213],"genre_scores_gemma":[0.7573309,0.0003081399,0.23747212,0.00021375727,0.00015052732,0.00022489806,0.0012988704,0.00012363786,0.002877246],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973731,0.00052421674,0.0002597634,0.0006592405,0.00091729825,0.00026634583],"domain_scores_gemma":[0.99107236,0.0039726417,0.001427154,0.000443819,0.0025219324,0.0005619943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030515871,0.00180501,0.0020075715,0.0063718157,0.0006515821,0.0019247028,0.0020144188,0.001394081,0.0018232196],"category_scores_gemma":[0.00894935,0.00067981734,0.001107847,0.0029113644,0.00038625376,0.0021121104,0.0015712596,0.0011204819,0.0006317442],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059745315,0.0015574874,0.05108267,0.00056010036,0.00055976765,0.00035676238,0.00032760034,0.31763616,0.0113454955,0.0031695885,0.007445448,0.6053615],"study_design_scores_gemma":[0.000015851498,0.00012299277,0.003572707,0.000023112612,0.000063430736,0.000043063454,0.000048244056,0.9926985,0.0011663045,0.001765344,0.00046106393,0.000019410701],"about_ca_topic_score_codex":0.008828976,"about_ca_topic_score_gemma":0.02033119,"teacher_disagreement_score":0.008828976,"about_ca_system_score_codex":0.0009816197,"about_ca_system_score_gemma":0.0031773928,"threshold_uncertainty_score":0.017555177},"labels":[],"label_agreement":null},{"id":"W4390411712","doi":"10.1007/s10664-023-10421-9","title":"Using knowledge units of programming languages to recommend reviewers for pull requests: an empirical study","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Baseline (sea); Operationalization; Code (set theory); Java; Task (project management); Recommender system; Artificial intelligence; Range (aeronautics); Code review; Information retrieval; Machine learning; Programming language; Natural language processing; Set (abstract data type); Software; Software quality; Software development; Engineering","score_opus":0.15121060831279526,"score_gpt":0.43717184664025377,"score_spread":0.2859612383274585,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390411712","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99503917,0.00020396752,0.0023077922,0.00013427397,0.000010870192,0.000160269,0.00017610771,0.00008822924,0.0018794374],"genre_scores_gemma":[0.99566627,0.00009293976,0.003308662,0.0000550349,0.00001848371,0.00008269205,0.00019712854,0.000033215674,0.0005454827],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.97778076,0.011807538,0.0018675673,0.0019828307,0.0058339746,0.00072736805],"domain_scores_gemma":[0.31930277,0.60268307,0.04427809,0.0098335715,0.018287364,0.005615154],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.017069582,0.0005901412,0.0006968158,0.0061153746,0.001292634,0.0044950945,0.0020345014,0.0019931498,0.003370466],"category_scores_gemma":[0.30234155,0.0005761931,0.000648469,0.0049046795,0.0012329669,0.0066303946,0.0015941915,0.0018257987,0.00079263846],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020956043,0.004528131,0.87652963,0.000817513,0.00040525172,0.00052688556,0.010525312,0.0020611912,0.0034155017,0.0008704147,0.0015042669,0.09672028],"study_design_scores_gemma":[0.00046785868,0.003999512,0.9033156,0.00039152315,0.0010880483,0.001487458,0.017254889,0.05477608,0.009449311,0.0032255244,0.0041977996,0.0003462907],"about_ca_topic_score_codex":0.0061583244,"about_ca_topic_score_gemma":0.006708993,"teacher_disagreement_score":0.9829304,"about_ca_system_score_codex":0.0016837134,"about_ca_system_score_gemma":0.0027424777,"threshold_uncertainty_score":0.09027362},"labels":[],"label_agreement":null},{"id":"W4390502905","doi":"10.1007/s10664-023-10362-3","title":"What is an app store? The software engineering perspective","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; University of Waterloo","funders":"European Commission","keywords":"App store; Computer science; World Wide Web; Mobile apps; Android (operating system); Software; Download; Operating system","score_opus":0.0255173147224049,"score_gpt":0.30159338532327223,"score_spread":0.27607607060086736,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390502905","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10865645,0.023769908,0.11352314,0.19521873,0.00080513826,0.00013022087,0.00084283913,0.00025577348,0.5567978],"genre_scores_gemma":[0.930356,0.017434454,0.021673748,0.0072508105,0.001619965,0.000095497795,0.00031311237,0.00014942892,0.021106988],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.99638605,0.0014447786,0.00019918606,0.00069292285,0.0008213468,0.0004556945],"domain_scores_gemma":[0.9848407,0.009982039,0.00090981415,0.0009550351,0.0020713417,0.001241039],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027386541,0.00097096816,0.0010793384,0.0066930284,0.004145586,0.02143852,0.0022405111,0.007065508,0.009774678],"category_scores_gemma":[0.011965342,0.0009070285,0.0006683621,0.0060154703,0.019485373,0.049226996,0.0031228645,0.004868644,0.0017176414],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002189354,0.00008518575,0.0018809543,0.00015062881,0.0000147070405,0.0002674933,0.0020050704,0.00046438043,0.00025640742,0.97710127,0.0031007521,0.014651353],"study_design_scores_gemma":[0.000010366011,0.000047212914,0.0019249951,0.00042499046,0.000036737667,0.00080966594,0.010916125,0.0035086332,0.0006585495,0.9331931,0.04843713,0.000032649616],"about_ca_topic_score_codex":0.011214853,"about_ca_topic_score_gemma":0.0080646,"teacher_disagreement_score":0.02143852,"about_ca_system_score_codex":0.0043008802,"about_ca_system_score_gemma":0.0042220703,"threshold_uncertainty_score":0.032699585},"labels":[],"label_agreement":null},{"id":"W4390749519","doi":"10.1007/s10664-023-10424-6","title":"Diversity in issue assignment: humans vs bots","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia; University of Waterloo","funders":"","keywords":"Diversity (politics); Open source; Process (computing); Open source software; Computer science; Ethnic group; Race (biology); White (mutation); Data science; World Wide Web; Knowledge management; Software; Political science; Sociology; Law","score_opus":0.02814452508204056,"score_gpt":0.2804815817334202,"score_spread":0.2523370566513796,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390749519","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9674597,0.0003876398,0.015055253,0.0013303076,0.00012620361,0.000112202004,0.00021667389,0.0001013738,0.015210705],"genre_scores_gemma":[0.99662054,0.000049483016,0.0014572095,0.00021190141,0.00005190174,0.000024479896,0.00008157063,0.000038822556,0.0014641053],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.97825456,0.013005136,0.000914752,0.003290963,0.0032717583,0.0012628153],"domain_scores_gemma":[0.8343616,0.1207037,0.0175813,0.016702753,0.0054600276,0.005190574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016549202,0.0004820863,0.00088210864,0.002817212,0.00204808,0.0039586565,0.001220461,0.0026049602,0.012094578],"category_scores_gemma":[0.14981082,0.000614248,0.00048645682,0.0015184684,0.003159627,0.008145609,0.0034599362,0.0028338553,0.0015962541],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003290029,0.0023725068,0.7301811,0.0006202207,0.00063036935,0.00078444486,0.020291531,0.010559322,0.010097382,0.050709043,0.012504265,0.15795979],"study_design_scores_gemma":[0.00066026923,0.0014689114,0.57055813,0.0003800274,0.00042181576,0.0031007498,0.02534705,0.099587075,0.0070546623,0.26958412,0.021538138,0.000299032],"about_ca_topic_score_codex":0.0016686138,"about_ca_topic_score_gemma":0.0020845742,"teacher_disagreement_score":0.016549202,"about_ca_system_score_codex":0.0006956828,"about_ca_system_score_gemma":0.0008275286,"threshold_uncertainty_score":0.08752155},"labels":[],"label_agreement":null},{"id":"W4391745357","doi":"10.1007/s10664-023-10437-1","title":"A study of common bug fix patterns in Rust","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Rust (programming language); Computer science; Compiler; Programming language; Software engineering","score_opus":0.03218305884190426,"score_gpt":0.3177540607064256,"score_spread":0.28557100186452133,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391745357","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9977331,0.00013645248,0.0010290731,0.000055833825,0.0000045118413,0.000018375904,0.00014182861,0.000048460857,0.0008322172],"genre_scores_gemma":[0.9972288,0.00006218439,0.0017525943,0.000024191415,0.000004135053,0.000016335796,0.00026068118,0.0000321706,0.00061873026],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9975048,0.00089597673,0.00019678826,0.00046707335,0.0007669172,0.00016852896],"domain_scores_gemma":[0.9384988,0.037400074,0.013607631,0.004722381,0.004408465,0.0013627138],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.00248267,0.00027496016,0.0003316651,0.0037046296,0.0007818677,0.00082215125,0.00080707046,0.0006708794,0.002519348],"category_scores_gemma":[0.034682855,0.00027829802,0.00045059103,0.0042866752,0.0007667778,0.0014669446,0.00089613075,0.0008551585,0.00034012456],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004437636,0.0008897392,0.8903564,0.00027733468,0.00025490372,0.0014854484,0.012496909,0.002118007,0.005062856,0.0029436252,0.0016110442,0.08205994],"study_design_scores_gemma":[0.00007938253,0.0012986964,0.9694408,0.00011802789,0.00014097827,0.0022801647,0.0069146925,0.011648963,0.002065659,0.0024875272,0.0034747494,0.00005034488],"about_ca_topic_score_codex":0.0039188885,"about_ca_topic_score_gemma":0.008626396,"teacher_disagreement_score":0.99751735,"about_ca_system_score_codex":0.0007726601,"about_ca_system_score_gemma":0.0008518915,"threshold_uncertainty_score":0.013129771},"labels":[],"label_agreement":null},{"id":"W4391835605","doi":"10.1007/s10664-024-10449-5","title":"Quantifying and characterizing clones of self-admitted technical debt in build systems","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Precursory Research for Embryonic Science and Technology; Japan Science and Technology Agency; Japan Society for the Promotion of Science","keywords":"Artifact (error); Technical debt; Computer science; Context (archaeology); Maintainability; Reliability (semiconductor); Code (set theory); Scale (ratio); Data science; Data mining; Software; Software engineering; Software development; Artificial intelligence; Programming language; Biology; Set (abstract data type); Geography","score_opus":0.029443787547574637,"score_gpt":0.29859674966661726,"score_spread":0.2691529621190426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391835605","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9918391,0.00008374062,0.007229241,0.000051053103,0.000003668952,0.00001970081,0.00007459055,0.00009160029,0.0006073612],"genre_scores_gemma":[0.9960174,0.000028632545,0.0035264408,0.000014349905,0.000004980561,0.000016616124,0.000120016186,0.00002932133,0.00024225352],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9951167,0.0012771062,0.00048699882,0.0009049158,0.001739077,0.0004752427],"domain_scores_gemma":[0.87102795,0.06320005,0.032885548,0.016513895,0.013592902,0.0027796132],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005711599,0.00041622657,0.0005845566,0.0031042118,0.0009772275,0.0029037127,0.0013624233,0.0015310316,0.0009089044],"category_scores_gemma":[0.08774076,0.0005703214,0.00041925604,0.0024384395,0.0016732505,0.004054311,0.0021511847,0.0012731774,0.00019429578],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001992014,0.0001832686,0.92622787,0.000094139876,0.00011906526,0.00043348392,0.0028852567,0.016506946,0.007964778,0.0084734615,0.000356133,0.036556475],"study_design_scores_gemma":[0.000039304214,0.00038272163,0.7431955,0.00012374127,0.00018627156,0.0012210222,0.003424992,0.20999956,0.014875297,0.024483563,0.0019807103,0.0000872324],"about_ca_topic_score_codex":0.0046688085,"about_ca_topic_score_gemma":0.005823241,"teacher_disagreement_score":0.005711599,"about_ca_system_score_codex":0.0020625128,"about_ca_system_score_gemma":0.0014660542,"threshold_uncertainty_score":0.030206203},"labels":[],"label_agreement":null},{"id":"W4391915512","doi":"10.1007/s10664-024-10443-x","title":"Studying the impact of risk assessment analytics on risk awareness and code review performance","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Waterloo","funders":"","keywords":"Analytics; Computer science; Data science; Code (set theory); Risk analysis (engineering); Business; Programming language","score_opus":0.04283266006473341,"score_gpt":0.3663655825837216,"score_spread":0.3235329225189882,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391915512","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99610823,0.00024549014,0.0016103714,0.00031242598,0.000015811345,0.000015418866,0.000053097698,0.00012711773,0.0015119911],"genre_scores_gemma":[0.99877626,0.00004219419,0.00077707536,0.000022359534,0.000011956332,0.000003858007,0.0000540276,0.000015214687,0.00029713794],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99318236,0.0031593712,0.0003971759,0.0008441944,0.0018131196,0.0006037511],"domain_scores_gemma":[0.6318019,0.31279323,0.028002288,0.009355239,0.013681233,0.0043661874],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.008920325,0.0005742394,0.00037110742,0.0018758987,0.00042927434,0.0025598418,0.00077103265,0.00092419406,0.0020455488],"category_scores_gemma":[0.14964607,0.00027729318,0.00047421193,0.0014029806,0.00064186525,0.0037879616,0.0007368459,0.0016614302,0.00041534012],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0042182803,0.005520944,0.6534511,0.0003757128,0.0009795391,0.000272735,0.0017011664,0.09593758,0.016034411,0.0043382924,0.0035333969,0.21363679],"study_design_scores_gemma":[0.00016381327,0.0067506833,0.4098689,0.00013157821,0.0006929798,0.00037022756,0.0028930593,0.5476822,0.022240875,0.0071008825,0.0019394966,0.00016521149],"about_ca_topic_score_codex":0.004915294,"about_ca_topic_score_gemma":0.005076402,"teacher_disagreement_score":0.9910797,"about_ca_system_score_codex":0.0012397529,"about_ca_system_score_gemma":0.002392359,"threshold_uncertainty_score":0.047175765},"labels":[],"label_agreement":null},{"id":"W4391998095","doi":"10.1007/s10664-023-10433-5","title":"Evaluating the impact of flaky simulators on testing autonomous driving systems","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Test (biology); Replication (statistics); Key (lock); Scope (computer science); Code coverage; Simulation; Machine learning; Operating system; Statistics; Software; Mathematics; Programming language","score_opus":0.08166117695653764,"score_gpt":0.382740428437695,"score_spread":0.3010792514811574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391998095","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99214363,0.00009029847,0.006487836,0.00007151493,0.000010248334,0.000053098927,0.00006417737,0.00018936192,0.00088982336],"genre_scores_gemma":[0.9934149,0.000027934047,0.0063054026,0.00001136169,0.0000020308896,0.000014734084,0.00005856738,0.000017168848,0.00014796472],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949168,0.0033146874,0.0002769813,0.0003475549,0.0008796799,0.00026438714],"domain_scores_gemma":[0.8266977,0.15567952,0.0047927215,0.0060870154,0.0053761364,0.0013670385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049743736,0.0008984674,0.0002683462,0.0011461416,0.0003428161,0.0005359555,0.0014810615,0.0010144076,0.0013395605],"category_scores_gemma":[0.084990814,0.00043974924,0.00033722742,0.0007095672,0.00084326364,0.0016840782,0.0007400524,0.0007697117,0.00013683004],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004246696,0.006938372,0.09320131,0.0005485198,0.00034047893,0.00031174027,0.0011181594,0.6489603,0.021025112,0.0026877995,0.0009309366,0.21969053],"study_design_scores_gemma":[0.0004684874,0.010175441,0.035986368,0.000088847206,0.0002204018,0.00018555214,0.0007044156,0.9236166,0.02551823,0.0022080604,0.00076826167,0.000059320566],"about_ca_topic_score_codex":0.0073980405,"about_ca_topic_score_gemma":0.010111958,"teacher_disagreement_score":0.0073980405,"about_ca_system_score_codex":0.0012165038,"about_ca_system_score_gemma":0.0012222741,"threshold_uncertainty_score":0.026307344},"labels":[],"label_agreement":null},{"id":"W4392224127","doi":"10.1007/s10664-024-10474-4","title":"An empirical study of challenges in machine learning asset management","year":2024,"lang":"en","type":"preprint","venue":"Empirical Software Engineering","topic":"Big Data and Business Intelligence","field":"Business, Management and Accounting","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Asset management; Empirical research; Asset (computer security); Business; Computer science; Knowledge management; Finance; Computer security; Mathematics; Statistics","score_opus":0.11135924812227603,"score_gpt":0.3433271423296989,"score_spread":0.23196789420742286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392224127","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98338175,0.0004090977,0.0025643024,0.003012372,0.000023817815,0.000046138368,0.00006537502,0.0000071554027,0.010490064],"genre_scores_gemma":[0.9984995,0.00009003956,0.00068635965,0.00009540749,0.000015396006,0.000016796923,0.00004361172,0.000004052918,0.00054887886],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99624157,0.0022318154,0.00018341567,0.00025453384,0.00086881575,0.00021984665],"domain_scores_gemma":[0.7948788,0.17925365,0.010718027,0.004714223,0.007774608,0.0026606594],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0071564126,0.00022115369,0.00029270622,0.0008153446,0.0014232316,0.0032930109,0.00091991515,0.0013376233,0.0032277037],"category_scores_gemma":[0.12584531,0.00021385655,0.00016566794,0.0020427676,0.0020887612,0.007206983,0.0014878481,0.0024238718,0.00032998092],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088242406,0.0032104647,0.65187156,0.00041199307,0.0001001848,0.00105373,0.02156719,0.013228659,0.0014266835,0.18224536,0.01074119,0.1132606],"study_design_scores_gemma":[0.00017761863,0.0012060887,0.46053115,0.0005249543,0.00008074605,0.0012042979,0.08858345,0.11896303,0.0024070803,0.29048085,0.03573503,0.00010574284],"about_ca_topic_score_codex":0.0031768868,"about_ca_topic_score_gemma":0.0034871341,"teacher_disagreement_score":0.0071564126,"about_ca_system_score_codex":0.0010976959,"about_ca_system_score_gemma":0.0017268873,"threshold_uncertainty_score":0.03784722},"labels":[],"label_agreement":null},{"id":"W4396610513","doi":"10.1007/s10664-024-10462-8","title":"Patterns of multi-container composition for service orchestration with Docker Compose","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Service-Oriented Architecture and Web Services","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Orchestration; Container (type theory); Service composition; Computer science; Service (business); Operating system; Business; Engineering; Marketing","score_opus":0.018755243539334786,"score_gpt":0.2640178082306203,"score_spread":0.24526256469128555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396610513","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2945208,0.00012259498,0.6713054,0.00052044593,0.00009923217,0.00043152322,0.0004210455,0.0058204415,0.026758537],"genre_scores_gemma":[0.6657634,0.00008618665,0.32247055,0.00007044097,0.000015486687,0.00021045179,0.0007969811,0.0010184594,0.009568084],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99600035,0.0010844053,0.0003975404,0.0006710475,0.0013756827,0.0004710698],"domain_scores_gemma":[0.9912344,0.002562038,0.0007713893,0.0037897702,0.0011321659,0.0005102103],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031084027,0.00039180994,0.0004414469,0.001416659,0.0014808063,0.00315074,0.0014096443,0.0011974096,0.005235153],"category_scores_gemma":[0.014608667,0.0005713317,0.0009025369,0.0016318433,0.001514852,0.0029226732,0.0019528144,0.0011375314,0.0015795323],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013127676,0.0010078334,0.11250703,0.00060070906,0.00026328943,0.0034021584,0.011895106,0.03908633,0.063042596,0.36304885,0.014875004,0.3889584],"study_design_scores_gemma":[0.00017036921,0.00045083175,0.045298103,0.00030499653,0.00028814736,0.005099931,0.0051348787,0.48608184,0.090031534,0.25786275,0.10904643,0.00023027441],"about_ca_topic_score_codex":0.004026605,"about_ca_topic_score_gemma":0.005421724,"teacher_disagreement_score":0.005235153,"about_ca_system_score_codex":0.0007823484,"about_ca_system_score_gemma":0.0017533404,"threshold_uncertainty_score":0.017513394},"labels":[],"label_agreement":null},{"id":"W4399259577","doi":"10.1007/s10664-024-10464-6","title":"Towards graph-anonymization of software analytics data: empirical study on JIT defect prediction","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Analytics; Data mining; Empirical research; Software; Graph; Data science; Statistics; Theoretical computer science; Mathematics; Programming language","score_opus":0.06589437020513635,"score_gpt":0.3439454061040775,"score_spread":0.27805103589894115,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399259577","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.70437276,0.00087839214,0.2645387,0.002258876,0.00026286443,0.00048381768,0.019300414,0.0032943212,0.0046098386],"genre_scores_gemma":[0.90428746,0.00038694777,0.07068875,0.0001752918,0.00010550605,0.0001785222,0.022695696,0.00019465364,0.0012872857],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9906985,0.0048358906,0.00048635417,0.0016229973,0.0019236414,0.00043268062],"domain_scores_gemma":[0.9258142,0.03109038,0.00780481,0.029469807,0.0050666747,0.00075417134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052941553,0.00049510226,0.0005464179,0.0045562177,0.0010566104,0.0019705785,0.0012283917,0.0012836949,0.0012443928],"category_scores_gemma":[0.053638384,0.00026905094,0.00066466106,0.00573226,0.001325302,0.0048141163,0.002234711,0.0017562087,0.0008564665],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012759748,0.0017373053,0.31383947,0.0012168259,0.0005604559,0.0008581134,0.005413889,0.11060078,0.016827654,0.0606988,0.047514793,0.43945596],"study_design_scores_gemma":[0.000102260274,0.0003318764,0.10062726,0.00032596735,0.00024648264,0.0013425812,0.004372085,0.6858599,0.021262916,0.14234181,0.043061703,0.00012513093],"about_ca_topic_score_codex":0.0036818672,"about_ca_topic_score_gemma":0.0038134577,"teacher_disagreement_score":0.0052941553,"about_ca_system_score_codex":0.00089730433,"about_ca_system_score_gemma":0.0020495018,"threshold_uncertainty_score":0.027998507},"labels":[],"label_agreement":null},{"id":"W4399333905","doi":"10.1007/s10664-024-10485-1","title":"A large-scale exploratory study on the proxy pattern in Ethereum","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Proxy (statistics); Scale (ratio); Computer science; Geography; Cartography; Machine learning","score_opus":0.026902270609266733,"score_gpt":0.2671268061620694,"score_spread":0.24022453555280268,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399333905","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9986739,0.000037902668,0.00032761713,0.00007007308,0.0000018853556,0.000022818735,0.00014848143,0.0000080756945,0.0007092752],"genre_scores_gemma":[0.9980123,0.000046964153,0.0006189359,0.000072749084,0.00000810329,0.000025793646,0.00043013072,0.000013804342,0.0007712235],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9979032,0.0010966151,0.0001340632,0.00021279119,0.0004420335,0.00021121014],"domain_scores_gemma":[0.960785,0.025942724,0.0062662987,0.00249535,0.0033674396,0.0011431586],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003084642,0.00020254868,0.00026604682,0.0025419476,0.0013052977,0.0010818616,0.0005612567,0.0008530067,0.0017328869],"category_scores_gemma":[0.028631112,0.00018823502,0.0001389376,0.0028870555,0.00096477225,0.0022796493,0.0011613413,0.0008566558,0.00064688077],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002707955,0.0011381595,0.932951,0.00015954295,0.00004641242,0.00088994903,0.021084275,0.00041573585,0.0025102105,0.0032399301,0.0026055267,0.034688365],"study_design_scores_gemma":[0.000022359436,0.00031150345,0.9617818,0.000091782764,0.0000270607,0.00080963137,0.023033487,0.0039864555,0.0019955859,0.0008782915,0.007031001,0.000031108793],"about_ca_topic_score_codex":0.0055811238,"about_ca_topic_score_gemma":0.009393797,"teacher_disagreement_score":0.0055811238,"about_ca_system_score_codex":0.0008163358,"about_ca_system_score_gemma":0.00062923273,"threshold_uncertainty_score":0.016313374},"labels":[],"label_agreement":null},{"id":"W4399334258","doi":"10.1007/s10664-024-10493-1","title":"Towards understanding barriers and mitigation strategies of software engineers with non-traditional educational and occupational backgrounds","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Disability Education and Employment","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia; University of Waterloo","funders":"","keywords":"Engineering; Software; Computer science; Knowledge management; Software engineering; Data science; Engineering management","score_opus":0.055875200237333236,"score_gpt":0.3262491438988426,"score_spread":0.27037394366150935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399334258","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9838266,0.0006489533,0.0036345697,0.0032398414,0.000017884679,0.00011801661,0.00011009027,0.000013851574,0.0083903],"genre_scores_gemma":[0.99648356,0.0003528766,0.0018321512,0.00020763448,0.0000033843276,0.00007510783,0.000060510938,0.000004715705,0.0009801876],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.99581015,0.001735696,0.0002283387,0.00042091642,0.0006471192,0.001157783],"domain_scores_gemma":[0.9781917,0.011859281,0.0045673745,0.00069425016,0.0030282233,0.0016591395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005834626,0.00046150907,0.0004600574,0.002690524,0.0022613402,0.005267716,0.0016095871,0.0019203129,0.005024643],"category_scores_gemma":[0.03107163,0.0004342677,0.00053140806,0.0018303083,0.0015894459,0.0077137067,0.0037349341,0.0019771417,0.0005064517],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001524372,0.0021076035,0.71657157,0.00074150355,0.00012582963,0.0004229052,0.16598666,0.0008911071,0.0013819687,0.028893229,0.0016530736,0.08107214],"study_design_scores_gemma":[0.000025654934,0.0002443803,0.4016707,0.0019984697,0.00012526616,0.00028183596,0.56425613,0.0033451477,0.0009408492,0.017423688,0.009635215,0.000052691237],"about_ca_topic_score_codex":0.02933346,"about_ca_topic_score_gemma":0.04054666,"teacher_disagreement_score":0.02933346,"about_ca_system_score_codex":0.0026146178,"about_ca_system_score_gemma":0.011722731,"threshold_uncertainty_score":0.05832547},"labels":[],"label_agreement":null},{"id":"W4399362212","doi":"10.1007/s10664-024-10448-6","title":"VulNet: Towards improving vulnerability management in the Maven ecosystem","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Concordia University","funders":"Indian Institute of Technology Gandhinagar","keywords":"Vulnerability (computing); Ecosystem; Environmental resource management; Environmental science; Business; Computer science; Ecology; Computer security; Biology","score_opus":0.01655092419775144,"score_gpt":0.2847365155113277,"score_spread":0.2681855913135763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399362212","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5720704,0.0025767505,0.3521417,0.004218999,0.00045293022,0.00036546445,0.0018963681,0.042530093,0.023747208],"genre_scores_gemma":[0.77748245,0.0010489348,0.20856154,0.0005584888,0.00012481575,0.00015954483,0.0028126591,0.002018018,0.007233575],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9984659,0.00062823476,0.00006375218,0.0002580002,0.00043394996,0.00015011792],"domain_scores_gemma":[0.99530715,0.001719514,0.00061060715,0.0012075295,0.00084714766,0.00030809228],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031429108,0.0010391893,0.0005337913,0.0026060469,0.0006237785,0.0022876542,0.0011293159,0.0009128672,0.002797179],"category_scores_gemma":[0.014732683,0.00033941423,0.0005105851,0.00077725196,0.00078150234,0.0046271454,0.0026214535,0.0014267002,0.00080001],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072669704,0.0010360362,0.088273585,0.0007392082,0.00056301325,0.00034932594,0.0012484534,0.10360704,0.028299343,0.035087198,0.03676631,0.70330375],"study_design_scores_gemma":[0.00014712223,0.0006443279,0.026060885,0.00033553847,0.0003288373,0.0007594572,0.000822234,0.8228072,0.028344076,0.054545224,0.06508914,0.000116031726],"about_ca_topic_score_codex":0.0028870213,"about_ca_topic_score_gemma":0.0053164596,"teacher_disagreement_score":0.0031429108,"about_ca_system_score_codex":0.0007032637,"about_ca_system_score_gemma":0.0018531017,"threshold_uncertainty_score":0.01662147},"labels":[],"label_agreement":null},{"id":"W4399362314","doi":"10.1007/s10664-024-10487-z","title":"Characterizing and classifying developer forum posts with their intentions","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Mitacs","keywords":"Computer science; Knowledge management","score_opus":0.025739535136037356,"score_gpt":0.2690836191399585,"score_spread":0.24334408400392113,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399362314","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9913147,0.00021431618,0.0035382276,0.00013428777,0.00007531432,0.0000620379,0.0010853799,0.00020678488,0.0033689686],"genre_scores_gemma":[0.99105495,0.00013090784,0.0039808718,0.000040213712,0.00011940046,0.00007856152,0.0020575013,0.00006130649,0.002476255],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9972192,0.0007718135,0.00029157868,0.00035885934,0.0010066524,0.0003518907],"domain_scores_gemma":[0.9407309,0.03978971,0.008013896,0.0022183405,0.0066301352,0.0026169477],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003196543,0.0005254932,0.0002942241,0.0067015933,0.0007252148,0.0017425946,0.0003581383,0.0009218437,0.0016820993],"category_scores_gemma":[0.03530728,0.0002213058,0.00033247017,0.003117072,0.0003692267,0.0023264594,0.0011114563,0.0007874074,0.001059538],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055288034,0.00043291744,0.8913572,0.00031687148,0.00009096902,0.00028412763,0.0029757188,0.00062797393,0.016032238,0.0011200089,0.004858622,0.081350386],"study_design_scores_gemma":[0.000035236535,0.00064891647,0.9508951,0.00012271707,0.00014478633,0.0006607403,0.005017277,0.024967857,0.006279759,0.0019326214,0.0092207985,0.00007423537],"about_ca_topic_score_codex":0.0013656802,"about_ca_topic_score_gemma":0.0037126583,"teacher_disagreement_score":0.0067015933,"about_ca_system_score_codex":0.00031044116,"about_ca_system_score_gemma":0.0005903927,"threshold_uncertainty_score":0.016905129},"labels":[],"label_agreement":null},{"id":"W4399615587","doi":"10.1007/s10664-024-10450-y","title":"How far are we with automated machine learning? characterization and challenges of AutoML toolkits","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Characterization (materials science); Computer science; Artificial intelligence; Machine learning; Nanotechnology; Materials science","score_opus":0.02592931027558718,"score_gpt":0.24558559859600892,"score_spread":0.21965628832042175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399615587","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12992011,0.015882432,0.55343443,0.22628555,0.00073083333,0.00026739592,0.0017167641,0.012267617,0.05949492],"genre_scores_gemma":[0.5914972,0.005695863,0.37214142,0.009213519,0.0009189151,0.0005719244,0.0024182657,0.006259265,0.011283677],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.93855,0.033181746,0.0034306245,0.0056978506,0.015436,0.003703799],"domain_scores_gemma":[0.73446995,0.14413506,0.0076879193,0.073461995,0.0312546,0.008990596],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.060150046,0.0010285215,0.0018175816,0.0063736998,0.0035670416,0.02285809,0.007983534,0.0059344466,0.01066951],"category_scores_gemma":[0.16444503,0.0018291858,0.0013605542,0.0052768094,0.015301138,0.05965002,0.015033552,0.011062662,0.008584305],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035148993,0.00033874292,0.020341834,0.0006984515,0.00007919706,0.00021302114,0.0046579125,0.0053306916,0.0020127269,0.5948865,0.028853128,0.3422363],"study_design_scores_gemma":[0.000058874928,0.00011744195,0.005944434,0.001011871,0.000036228543,0.000702321,0.0051488,0.054569095,0.0037807196,0.78648853,0.1419605,0.00018113166],"about_ca_topic_score_codex":0.0044885892,"about_ca_topic_score_gemma":0.0040683174,"teacher_disagreement_score":0.93985,"about_ca_system_score_codex":0.0039807474,"about_ca_system_score_gemma":0.008138442,"threshold_uncertainty_score":0.3181076},"labels":[],"label_agreement":null},{"id":"W4399644038","doi":"10.1007/s10664-024-10457-5","title":"Utilization of pre-trained language models for adapter-based knowledge transfer in software engineering","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Okanagan University College; University of British Columbia, Okanagan Campus; University of British Columbia","funders":"","keywords":"Adapter (computing); Computer science; Code (set theory); Automatic summarization; Downstream (manufacturing); Software; Artificial intelligence; Natural language processing; Source code; Programming language; Engineering; Computer hardware","score_opus":0.046160511992757766,"score_gpt":0.31951923156039336,"score_spread":0.2733587195676356,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399644038","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32993022,0.00051434815,0.637453,0.0005396535,0.00032052258,0.00062240934,0.0009771938,0.018688751,0.010953998],"genre_scores_gemma":[0.8385911,0.00025424233,0.1538385,0.00029415224,0.00004879279,0.0004925391,0.0021710827,0.000576992,0.0037325718],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980392,0.00081568153,0.00016980113,0.0005940477,0.00023328981,0.00014806434],"domain_scores_gemma":[0.985067,0.0092957895,0.00041206923,0.002288898,0.0026167247,0.00031955453],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037321278,0.001343607,0.0007218871,0.0014722736,0.0006353962,0.0020035126,0.0021061231,0.0014426516,0.004733868],"category_scores_gemma":[0.024499623,0.0006136435,0.0009057356,0.0010398434,0.0005029205,0.006184866,0.0034896436,0.002789681,0.0031158703],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008343216,0.0013152974,0.010721737,0.00041529493,0.00019983425,0.00045292376,0.0016617961,0.0588063,0.029607778,0.0034365996,0.006341578,0.88620645],"study_design_scores_gemma":[0.000083711624,0.0004824591,0.003448649,0.00010598523,0.000198966,0.0002991454,0.00065086153,0.9351079,0.04640855,0.0091023715,0.004035414,0.00007592981],"about_ca_topic_score_codex":0.00638869,"about_ca_topic_score_gemma":0.0067452933,"teacher_disagreement_score":0.00638869,"about_ca_system_score_codex":0.0010652762,"about_ca_system_score_gemma":0.00225697,"threshold_uncertainty_score":0.019737601},"labels":[],"label_agreement":null},{"id":"W4399686980","doi":"10.1007/s10664-024-10500-5","title":"Common challenges of deep reinforcement learning applications development: an empirical study","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de recherche du Québec; Canadian Institute for Advanced Research","keywords":"Computer science; Taxonomy (biology); Leverage (statistics); Popularity; Reinforcement learning; Artificial intelligence; Data science; Software engineering","score_opus":0.039303453045082565,"score_gpt":0.3277676583422513,"score_spread":0.28846420529716876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399686980","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.987041,0.00031186646,0.0067943605,0.0009836364,0.000018697485,0.0001066572,0.00008206598,0.000091182636,0.00457047],"genre_scores_gemma":[0.9960871,0.00010104471,0.002775226,0.0000744711,0.000007262242,0.00004046179,0.00008242083,0.000022175567,0.00080985343],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9903364,0.0044991337,0.0006255725,0.0010198565,0.0029899962,0.0005290202],"domain_scores_gemma":[0.82047254,0.13490298,0.013224609,0.012646364,0.014958517,0.0037949746],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.010790749,0.0003709958,0.00035440119,0.0010528815,0.0011102257,0.002402581,0.0013159121,0.0015164483,0.002389452],"category_scores_gemma":[0.12243964,0.0003613032,0.00028923724,0.0012688949,0.0014179333,0.004202162,0.0022460246,0.0023739596,0.00055567373],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018156403,0.006771544,0.48493907,0.00096737366,0.00018547906,0.0012108996,0.015945604,0.030475266,0.0074069574,0.02877593,0.010908402,0.41059783],"study_design_scores_gemma":[0.00036116663,0.004829712,0.3836811,0.0009696836,0.00024451723,0.0025596744,0.030809632,0.4510855,0.016185759,0.06619454,0.042821106,0.00025769163],"about_ca_topic_score_codex":0.0026124239,"about_ca_topic_score_gemma":0.0035328737,"teacher_disagreement_score":0.98920923,"about_ca_system_score_codex":0.0017315651,"about_ca_system_score_gemma":0.0025650982,"threshold_uncertainty_score":0.057067633},"labels":[],"label_agreement":null},{"id":"W4399703649","doi":"10.1007/s10664-024-10492-2","title":"Post deployment recycling of machine learning models","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Software deployment; Reuse; Computer science; Artificial intelligence; Machine learning; Baseline (sea); Artificial neural network; Inference; Random forest; Logistic regression; Predictive modelling; Engineering","score_opus":0.03166989974971166,"score_gpt":0.27371393124768234,"score_spread":0.24204403149797069,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399703649","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.620885,0.0015485188,0.2914445,0.006705656,0.0014509815,0.0007016083,0.0024423627,0.04900776,0.025813553],"genre_scores_gemma":[0.8229087,0.00054506445,0.13871571,0.00090771716,0.00020416714,0.0002198805,0.004343373,0.0062948843,0.02586046],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98965186,0.0024186862,0.0007781363,0.0011519593,0.004920952,0.0010783111],"domain_scores_gemma":[0.88542354,0.032728318,0.0026936885,0.057734743,0.02001512,0.0014045733],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010102768,0.0014898132,0.0014981099,0.0035041596,0.0013624226,0.0049114637,0.0031332402,0.0019021608,0.010044134],"category_scores_gemma":[0.104782,0.0016968126,0.001711015,0.0022739777,0.0017137293,0.0059718546,0.0045841313,0.0037600659,0.0061874134],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015658495,0.0018055418,0.037769888,0.00080087956,0.0003827734,0.0014026596,0.0032773677,0.0428395,0.03048959,0.023734894,0.045738284,0.81019276],"study_design_scores_gemma":[0.00015772237,0.0011934338,0.038096156,0.00051021605,0.00050227763,0.0018998072,0.0022422953,0.7388988,0.09912766,0.040330026,0.07684005,0.00020155098],"about_ca_topic_score_codex":0.008925815,"about_ca_topic_score_gemma":0.011659701,"teacher_disagreement_score":0.010102768,"about_ca_system_score_codex":0.0015119,"about_ca_system_score_gemma":0.003668542,"threshold_uncertainty_score":0.053429186},"labels":[],"label_agreement":null},{"id":"W4399774065","doi":"10.1007/s10664-024-10452-w","title":"A literature review and existing challenges on software logging practices","year":2024,"lang":"en","type":"review","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; École de Technologie Supérieure","funders":"","keywords":"Logging; Computer science; Software; Software engineering; Data science; Forestry; Geography; Operating system","score_opus":0.10665188973831913,"score_gpt":0.3761438582854184,"score_spread":0.2694919685470993,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399774065","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00048311162,0.99622464,0.00031312738,0.0017486085,0.00017791915,0.000012427612,0.00008786294,0.000010352157,0.00094192545],"genre_scores_gemma":[0.0027977622,0.995673,0.0005146549,0.0006170534,0.0001535946,0.000016789292,0.000078613215,0.0000042331285,0.00014432649],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9965754,0.00088088855,0.00085678895,0.0004712409,0.0010762744,0.00013946083],"domain_scores_gemma":[0.9423085,0.04616555,0.0039106123,0.000690451,0.0063377237,0.00058717135],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0070172856,0.000829611,0.0017538253,0.0145491585,0.00077791215,0.003670958,0.0015047073,0.0018738304,0.0047967876],"category_scores_gemma":[0.033174783,0.0005931231,0.00095349544,0.02040884,0.0015816287,0.00654263,0.0015971086,0.0021292707,0.0009371097],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007451426,0.000088836845,0.0021181975,0.11993442,0.00017013629,0.00018060567,0.000867004,0.00027359158,0.00029219151,0.0076817935,0.02585153,0.84246725],"study_design_scores_gemma":[0.000029031817,0.00012664551,0.010694066,0.34529987,0.0018091694,0.0009552638,0.0038698553,0.00030058535,0.00044632138,0.009501671,0.6268883,0.00007929719],"about_ca_topic_score_codex":0.0060256803,"about_ca_topic_score_gemma":0.013874956,"teacher_disagreement_score":0.9929827,"about_ca_system_score_codex":0.0023378392,"about_ca_system_score_gemma":0.012556219,"threshold_uncertainty_score":0.0371114},"labels":[],"label_agreement":null},{"id":"W4399871949","doi":"10.1007/s10664-024-10501-4","title":"Systematic Evaluation of Deep Learning Models for Log-based Failure Prediction","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"HORIZON EUROPE Framework Programme; Canada Research Chairs; Natural Sciences and Engineering Research Council of Canada; European Commission; Science Foundation Ireland; Université du Luxembourg","keywords":"Computer science; Machine learning; Artificial intelligence; Artificial neural network; Algorithm; Convolutional neural network; Deep learning; Encoder; Predictive modelling; Data mining","score_opus":0.030171465703690212,"score_gpt":0.2758273760452596,"score_spread":0.24565591034156936,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399871949","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.93164873,0.002117273,0.052036382,0.00070023537,0.00022404506,0.0002929029,0.0061582853,0.0044916533,0.0023305644],"genre_scores_gemma":[0.9628586,0.00028229898,0.025295153,0.00010565032,0.000029097457,0.00016424323,0.010546565,0.00007952136,0.0006387698],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.9984054,0.00072338857,0.00014841801,0.00037934852,0.00022905588,0.000114332],"domain_scores_gemma":[0.9931599,0.0042577093,0.00042014243,0.00088722876,0.0011240636,0.00015097416],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003726257,0.001768218,0.0007091052,0.0012423529,0.0003029644,0.0007141958,0.0018444265,0.0011100601,0.0012183262],"category_scores_gemma":[0.00959413,0.00050665694,0.0010962366,0.0007957489,0.00053309853,0.001513455,0.00095493405,0.0014350204,0.0004117301],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009275912,0.00067669555,0.018202465,0.0005151284,0.0003982491,0.000114806266,0.000053726366,0.8623008,0.0035086158,0.00074906374,0.004890249,0.107662536],"study_design_scores_gemma":[0.00003233395,0.00021548558,0.0020330085,0.000023197632,0.00003648755,0.000018129194,0.000017145692,0.9938117,0.0031805774,0.0003102134,0.0003123535,0.000009351405],"about_ca_topic_score_codex":0.011890234,"about_ca_topic_score_gemma":0.01218626,"teacher_disagreement_score":0.011890234,"about_ca_system_score_codex":0.0015603675,"about_ca_system_score_gemma":0.00095286017,"threshold_uncertainty_score":0.023642063},"labels":[],"label_agreement":null},{"id":"W4400289352","doi":"10.1007/s10664-024-10476-2","title":"Design smells in multi-language systems and bug-proneness: a survival analysis","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Natural language processing; Programming language; Linguistics; Artificial intelligence","score_opus":0.050847855513788974,"score_gpt":0.3200556375862844,"score_spread":0.26920778207249546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400289352","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99750525,0.00010983171,0.0018684545,0.0000458291,0.0000020335694,0.000006017773,0.000085185755,0.000038190006,0.00033919633],"genre_scores_gemma":[0.9992454,0.000023714108,0.00042479992,0.0000049325677,0.0000022825293,0.0000066778066,0.00007396958,0.000012009797,0.00020636791],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99825436,0.0006952458,0.0001424062,0.00028809457,0.00041620826,0.00020356034],"domain_scores_gemma":[0.89172286,0.0746526,0.020106262,0.00537378,0.006193399,0.0019511753],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006945815,0.00045381923,0.0004444632,0.0034781678,0.00056042866,0.001384672,0.00065695425,0.0008309903,0.0032929387],"category_scores_gemma":[0.04203862,0.00032597428,0.0015222736,0.002380495,0.0010659051,0.0023733093,0.0012226589,0.0014094951,0.0004726969],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003155752,0.00015771456,0.9744241,0.000052556403,0.00023338127,0.00022722031,0.001234919,0.0033241722,0.0017032181,0.00079228374,0.000180681,0.017354203],"study_design_scores_gemma":[0.000025228486,0.00085157884,0.9433787,0.000050198392,0.00030525902,0.00070285314,0.001658478,0.049065832,0.0016664536,0.001973433,0.0002798838,0.00004219166],"about_ca_topic_score_codex":0.0026380522,"about_ca_topic_score_gemma":0.0025579785,"teacher_disagreement_score":0.006945815,"about_ca_system_score_codex":0.0006648172,"about_ca_system_score_gemma":0.00067360606,"threshold_uncertainty_score":0.03673345},"labels":[],"label_agreement":null},{"id":"W4400289394","doi":"10.1007/s10664-024-10499-9","title":"What causes exceptions in machine learning applications? Mining machine learning-related stack traces on Stack Overflow","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Stack (abstract data type); Computer science; Call stack; Machine learning; Artificial intelligence; Operating system","score_opus":0.021904454347741253,"score_gpt":0.28656119998284973,"score_spread":0.2646567456351085,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400289394","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98572415,0.00041248178,0.0114408815,0.0004827423,0.00003970042,0.000035207588,0.0007174711,0.00043481356,0.000712569],"genre_scores_gemma":[0.99408406,0.00016957348,0.0047123255,0.000059266524,0.000025417703,0.000015851583,0.0006503188,0.000047398025,0.00023582205],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9974981,0.0004746332,0.0003037557,0.00051431137,0.0009394512,0.00026986332],"domain_scores_gemma":[0.96102065,0.023018928,0.007602537,0.0033537361,0.0038702523,0.0011339026],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027100567,0.00057482754,0.0005450023,0.0033875222,0.00061447313,0.0017882559,0.0009512736,0.0010611892,0.000674698],"category_scores_gemma":[0.05564891,0.00043840904,0.0005852902,0.0028782713,0.00064476655,0.0031378828,0.000797621,0.0017826485,0.00028144367],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042181346,0.00042640814,0.88351685,0.0002179122,0.00019699159,0.0009987882,0.001054506,0.011550759,0.0053466302,0.003251977,0.002336487,0.09068097],"study_design_scores_gemma":[0.000050747723,0.00039030658,0.45540196,0.0002967685,0.00037260333,0.0018244598,0.0026691891,0.47552213,0.016645046,0.042045467,0.00467832,0.000102916216],"about_ca_topic_score_codex":0.0054500476,"about_ca_topic_score_gemma":0.006796717,"teacher_disagreement_score":0.0054500476,"about_ca_system_score_codex":0.0006176682,"about_ca_system_score_gemma":0.001777465,"threshold_uncertainty_score":0.014332354},"labels":[],"label_agreement":null},{"id":"W4400608520","doi":"10.1007/s10664-024-10488-y","title":"An empirical study on cross-component dependent changes: A case study on the components of OpenStack","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec","funders":"","keywords":"Component (thermodynamics); Computer science; Physics","score_opus":0.0628725410747167,"score_gpt":0.35741987604312486,"score_spread":0.2945473349684082,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400608520","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99796814,0.0000310282,0.0008121802,0.00003381755,0.0000031794423,0.000034011253,0.000039761435,0.000019846308,0.0010579732],"genre_scores_gemma":[0.99736303,0.00004580902,0.0013479932,0.000022301252,0.000004340721,0.000024547573,0.00014997282,0.000026706331,0.0010152558],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99720204,0.0010989393,0.00017431012,0.0004076543,0.00082989153,0.00028721234],"domain_scores_gemma":[0.95251155,0.03213399,0.0039868206,0.004442829,0.00575159,0.0011733584],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033057372,0.00039135618,0.00030871457,0.0019082696,0.0014803755,0.0010936346,0.0012926082,0.0011683139,0.0022615164],"category_scores_gemma":[0.025844553,0.00033194633,0.0003502771,0.0024159693,0.0016430902,0.002808721,0.0012115539,0.0013101846,0.0005285383],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015583406,0.011789763,0.7396431,0.00086045125,0.00020190296,0.008160987,0.054222647,0.011162829,0.016370682,0.007206391,0.0041246023,0.1446983],"study_design_scores_gemma":[0.0001526867,0.0035570448,0.85029006,0.00019224176,0.00023500912,0.0043454,0.068157844,0.033247523,0.021230102,0.0026816523,0.015750417,0.00016002267],"about_ca_topic_score_codex":0.007268704,"about_ca_topic_score_gemma":0.010779171,"teacher_disagreement_score":0.007268704,"about_ca_system_score_codex":0.0014258844,"about_ca_system_score_gemma":0.0010010236,"threshold_uncertainty_score":0.017482579},"labels":[],"label_agreement":null},{"id":"W4400832950","doi":"10.1007/s10664-024-10497-x","title":"Does using Bazel help speed up continuous integration builds?","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Heuristics; Duration (music); Artifact (error); Scheduling (production processes); Service (business); Software engineering; Operating system; Engineering; Operations management; Artificial intelligence","score_opus":0.019175855886186262,"score_gpt":0.27795946949845585,"score_spread":0.2587836136122696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400832950","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8287362,0.0027609668,0.09552732,0.009018746,0.00075453747,0.00026676568,0.00046715024,0.015392431,0.047075875],"genre_scores_gemma":[0.93234,0.00053352077,0.060436863,0.00042241727,0.00009741399,0.00005539187,0.00029253907,0.00067757984,0.005144281],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9972481,0.0009142813,0.00020437156,0.00033061672,0.0008482033,0.0004543687],"domain_scores_gemma":[0.96226937,0.020863611,0.0042345175,0.0071990434,0.003843719,0.0015897359],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054774587,0.0006649953,0.00050037,0.0013602582,0.0006223394,0.00302286,0.0016314805,0.0015285901,0.01453758],"category_scores_gemma":[0.062190168,0.0005780424,0.00040329772,0.0009321256,0.0004621179,0.007219621,0.0012666585,0.0011474229,0.004983804],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015233035,0.0015560285,0.066345155,0.00052559027,0.00011639004,0.00019344811,0.0010300475,0.0092300465,0.019447826,0.008053269,0.011793717,0.88018525],"study_design_scores_gemma":[0.0022197298,0.005819476,0.20007944,0.001049883,0.00093522365,0.0027129475,0.006281517,0.35776064,0.14862698,0.05637937,0.21763027,0.000504639],"about_ca_topic_score_codex":0.0038793266,"about_ca_topic_score_gemma":0.0073955893,"teacher_disagreement_score":0.01453758,"about_ca_system_score_codex":0.00056205754,"about_ca_system_score_gemma":0.0018252139,"threshold_uncertainty_score":0.04863304},"labels":[],"label_agreement":null},{"id":"W4400988474","doi":"10.1007/s10664-024-10518-9","title":"The impact of concept drift and data leakage on log level prediction models","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Queen's University","funders":"","keywords":"Interpretability; Debugging; Computer science; Context (archaeology); Statement (logic); Machine learning; Data mining; Data science; Deep learning; Artificial intelligence; Programming language","score_opus":0.09240412971559399,"score_gpt":0.34139983353286124,"score_spread":0.24899570381726727,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400988474","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8076637,0.0030053363,0.18217303,0.0027193823,0.00036090682,0.000106692845,0.0013188405,0.0011785999,0.0014736236],"genre_scores_gemma":[0.987201,0.00032481007,0.0112857185,0.00015983451,0.000058772133,0.000025818057,0.0004268471,0.000064620195,0.00045257018],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98991686,0.004693016,0.0009642222,0.0014888915,0.0023012476,0.0006357379],"domain_scores_gemma":[0.7289177,0.23835848,0.008317854,0.014073838,0.009015996,0.0013161974],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.034569725,0.0009616722,0.0010651809,0.00143881,0.0008819599,0.0026702578,0.0012932122,0.0014256977,0.0012207255],"category_scores_gemma":[0.22803107,0.0006332552,0.0010683464,0.0016707988,0.0012585171,0.0059071756,0.001810634,0.0032545356,0.00029125618],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023054446,0.0005558642,0.23438098,0.00029080902,0.00053376495,0.00045579227,0.00043692006,0.5917906,0.0036013161,0.009135723,0.0026626473,0.1538501],"study_design_scores_gemma":[0.00001981174,0.00022029638,0.007744998,0.000035909445,0.000112429174,0.00020581383,0.00006198085,0.9827079,0.002111096,0.006445262,0.00031694313,0.00001749681],"about_ca_topic_score_codex":0.005098096,"about_ca_topic_score_gemma":0.0043852255,"teacher_disagreement_score":0.034569725,"about_ca_system_score_codex":0.0015226738,"about_ca_system_score_gemma":0.002685632,"threshold_uncertainty_score":0.18282437},"labels":[],"label_agreement":null},{"id":"W4401109317","doi":"10.1007/s10664-024-10523-y","title":"Dependabot and security pull requests: large empirical study","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Computer science; Empirical research; Computer security; Statistics; Mathematics","score_opus":0.014386359565617011,"score_gpt":0.30830945252597364,"score_spread":0.2939230929603566,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401109317","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99352115,0.0014039709,0.0011835137,0.0006119989,0.000025377452,0.0001159558,0.0015350863,0.000036070658,0.0015669074],"genre_scores_gemma":[0.99489117,0.0006446386,0.001215152,0.00039573706,0.000041570755,0.00022646236,0.0019571707,0.00005608213,0.0005719676],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98607475,0.0071985377,0.00094889075,0.0019294834,0.0029803156,0.00086805306],"domain_scores_gemma":[0.69867575,0.2408425,0.03522725,0.011096582,0.009669435,0.004488451],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016828645,0.0007262324,0.0007438652,0.0029908887,0.0016752151,0.0019663593,0.0020227626,0.001822547,0.005276366],"category_scores_gemma":[0.09258271,0.00080637285,0.0013991086,0.004500958,0.0023045912,0.0053416416,0.0026877553,0.006236482,0.0017486055],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019227342,0.00092740334,0.98138237,0.0002695883,0.00038319489,0.0002865956,0.0024954928,0.00033286153,0.00009475973,0.0005266117,0.0035295424,0.009579234],"study_design_scores_gemma":[0.000059027137,0.00030782056,0.9807194,0.00032614876,0.00029797116,0.00050551636,0.0072073387,0.003939812,0.00020587568,0.00068293826,0.0056959875,0.00005208133],"about_ca_topic_score_codex":0.009681081,"about_ca_topic_score_gemma":0.010422601,"teacher_disagreement_score":0.016828645,"about_ca_system_score_codex":0.0009951887,"about_ca_system_score_gemma":0.0014440027,"threshold_uncertainty_score":0.08899945},"labels":[],"label_agreement":null},{"id":"W4401253331","doi":"10.1007/s10664-024-10514-z","title":"IRJIT: A simple, online, information retrieval approach for just-in-time software defect prediction","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Alberta","funders":"","keywords":"Computer science; Simple (philosophy); Information retrieval; Data mining; Software; Machine learning; Artificial intelligence; Programming language","score_opus":0.02499642994780278,"score_gpt":0.28696472697798037,"score_spread":0.2619682970301776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401253331","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024229,0.0014116683,0.9048358,0.0002769655,0.0003829768,0.0005856165,0.0057246634,0.058323096,0.004230149],"genre_scores_gemma":[0.17708942,0.00081049284,0.800072,0.0003234922,0.00042125955,0.00055120914,0.010663247,0.0011435831,0.008925277],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977552,0.000421977,0.00020515954,0.0003883019,0.0010816557,0.00014773577],"domain_scores_gemma":[0.9958358,0.0017591261,0.00038779984,0.0009405782,0.0009013762,0.00017522903],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002178702,0.002038982,0.002214526,0.0060680304,0.0006905059,0.0020196203,0.0033612442,0.0017824145,0.006242049],"category_scores_gemma":[0.009583185,0.00051715784,0.0014347476,0.0037278612,0.0004088961,0.00387443,0.0019290586,0.001532655,0.0062350156],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007810759,0.0010545466,0.0058878404,0.0007455108,0.00036831884,0.0003364351,0.00012450937,0.014620537,0.02048379,0.0041927164,0.0443754,0.90702933],"study_design_scores_gemma":[0.00019887743,0.00079149334,0.005880931,0.0000747267,0.00029111482,0.00089470146,0.000121027406,0.9249217,0.031626076,0.014328602,0.020663936,0.0002067572],"about_ca_topic_score_codex":0.0039370563,"about_ca_topic_score_gemma":0.0064925184,"teacher_disagreement_score":0.006242049,"about_ca_system_score_codex":0.00050859334,"about_ca_system_score_gemma":0.0014879048,"threshold_uncertainty_score":0.020881712},"labels":[],"label_agreement":null},{"id":"W4401667316","doi":"10.1007/s10664-024-10533-w","title":"Impact of log parsing on deep learning-based anomaly detection","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg; Science Foundation Ireland","keywords":"Anomaly detection; Parsing; Anomaly (physics); Computer science; Artificial intelligence; Natural language processing; Deep learning; Physics","score_opus":0.012061831520582988,"score_gpt":0.2753947141403206,"score_spread":0.2633328826197376,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401667316","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8435624,0.00369735,0.1222744,0.0024950188,0.0003532746,0.00016612936,0.0028651906,0.020284854,0.0043013818],"genre_scores_gemma":[0.9526482,0.00041108835,0.04151486,0.00032795485,0.00006746215,0.000055887085,0.0037867464,0.0003352509,0.0008525466],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9897361,0.0039274925,0.00091056875,0.0018205825,0.0029930992,0.00061215676],"domain_scores_gemma":[0.9155621,0.06316587,0.0049560866,0.0099450825,0.0055105397,0.00086032256],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01030849,0.0018583243,0.0009888668,0.002536975,0.0007560004,0.0022571632,0.0018016032,0.0014104681,0.0007650925],"category_scores_gemma":[0.06600802,0.00049149554,0.0009614868,0.0021719807,0.0010738571,0.005159288,0.0017983827,0.0033342405,0.0006541381],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015810694,0.0017216303,0.20131151,0.0006682039,0.0006002613,0.000399065,0.0004394679,0.25584757,0.0089599015,0.0030913162,0.01717904,0.5082008],"study_design_scores_gemma":[0.00004570167,0.0003842047,0.023744233,0.000096326745,0.00013901218,0.0003056887,0.00020216126,0.95175654,0.012663083,0.0074854195,0.0031134284,0.00006424876],"about_ca_topic_score_codex":0.0062546683,"about_ca_topic_score_gemma":0.0063299085,"teacher_disagreement_score":0.01030849,"about_ca_system_score_codex":0.0013313009,"about_ca_system_score_gemma":0.0020216694,"threshold_uncertainty_score":0.05451715},"labels":[],"label_agreement":null},{"id":"W4401995539","doi":"10.1007/s10664-024-10535-8","title":"Data-access performance anti-patterns in data-intensive systems","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; NoSQL; Data access; Data quality; Database; Component (thermodynamics); Scalability; Engineering","score_opus":0.09091578223945451,"score_gpt":0.3403247534093067,"score_spread":0.24940897116985217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401995539","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9953312,0.00007145715,0.0033568884,0.00022033454,0.000009069106,0.00000899509,0.000118530355,0.00010268302,0.00078080525],"genre_scores_gemma":[0.9992951,0.000010845024,0.00046885258,0.000014809727,0.000004853656,0.0000035588016,0.000049461138,0.000011210944,0.00014127127],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9973404,0.00079913886,0.0002762301,0.00048622588,0.00082060957,0.0002775157],"domain_scores_gemma":[0.9051119,0.05576882,0.019720245,0.008911435,0.009142757,0.0013449116],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026143803,0.0002008735,0.00032871804,0.001149871,0.0004871387,0.0012531464,0.00081598386,0.00059887394,0.0014187179],"category_scores_gemma":[0.05276729,0.0003918779,0.00020532304,0.0012978171,0.00071221415,0.0019755352,0.0007099217,0.001049831,0.00033452376],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009469747,0.00052300334,0.9201024,0.0001246414,0.000116300274,0.0001789043,0.00082507444,0.009549984,0.010606145,0.0054297373,0.0011307402,0.050466057],"study_design_scores_gemma":[0.000055887147,0.00084672566,0.751879,0.000052364005,0.000106196865,0.0010124281,0.0013783696,0.20900747,0.020924825,0.013199026,0.0014950688,0.00004267411],"about_ca_topic_score_codex":0.0021393734,"about_ca_topic_score_gemma":0.002410044,"teacher_disagreement_score":0.0026143803,"about_ca_system_score_codex":0.00065744994,"about_ca_system_score_gemma":0.0008467403,"threshold_uncertainty_score":0.013826311},"labels":[],"label_agreement":null},{"id":"W4402050968","doi":"10.1007/s10664-024-10477-1","title":"On combining commit grouping and build skip prediction to reduce redundant continuous integration activity","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Commit; Computer science; Database","score_opus":0.021112466439620347,"score_gpt":0.28530239328041973,"score_spread":0.2641899268407994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402050968","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1718079,0.0015056642,0.80555606,0.0012571798,0.0002403689,0.00027324577,0.0006288576,0.014174655,0.004556014],"genre_scores_gemma":[0.56350946,0.00034041377,0.4304113,0.00043854624,0.00012338156,0.00011979702,0.0014587725,0.0005792163,0.0030191133],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973877,0.00072152045,0.00015177742,0.0006156014,0.00084978685,0.00027353855],"domain_scores_gemma":[0.9879159,0.005293795,0.0007641556,0.0036806841,0.001984609,0.00036085243],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022525704,0.0014309543,0.0015332169,0.0024642586,0.000890379,0.0013739595,0.0026095342,0.0012701324,0.0028831777],"category_scores_gemma":[0.012861779,0.0006032662,0.000674581,0.0023650615,0.00069336087,0.0036562039,0.002183112,0.0017686051,0.0010219339],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006268266,0.0010995874,0.018312886,0.0001777127,0.00013680117,0.000117930766,0.00021639712,0.06797091,0.01682017,0.0030859609,0.009734232,0.8817005],"study_design_scores_gemma":[0.00009877834,0.00038937965,0.006435947,0.000051818253,0.00015137455,0.00013259174,0.0001725056,0.969964,0.010675444,0.009599287,0.0022880277,0.000040778297],"about_ca_topic_score_codex":0.008296199,"about_ca_topic_score_gemma":0.020324666,"teacher_disagreement_score":0.008296199,"about_ca_system_score_codex":0.00052424043,"about_ca_system_score_gemma":0.0028789435,"threshold_uncertainty_score":0.016495824},"labels":[],"label_agreement":null},{"id":"W4402136389","doi":"10.1007/s10664-024-10528-7","title":"Consensus task interaction trace recommender to guide developers’ software navigation","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Concordia University; Polytechnique Montréal","funders":"","keywords":"TRACE (psycholinguistics); Recommender system; Computer science; Task (project management); Software; World Wide Web; Human–computer interaction; Software engineering; Data science; Engineering; Programming language; Systems engineering","score_opus":0.036349133622559676,"score_gpt":0.32891704916676656,"score_spread":0.2925679155442069,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402136389","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22719708,0.0013764851,0.7227247,0.0013858497,0.00048140137,0.00049479207,0.002355515,0.027004274,0.016979935],"genre_scores_gemma":[0.78124505,0.0002731771,0.19527408,0.00033873584,0.00012208255,0.00027691646,0.0041349083,0.0009917214,0.017343342],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971468,0.0010019208,0.00017364146,0.00052419334,0.0009747117,0.00017875066],"domain_scores_gemma":[0.97887635,0.008946027,0.0010898497,0.0030044,0.0069270073,0.0011563293],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003049264,0.0012030613,0.0008636453,0.0039352295,0.0010446034,0.0016287094,0.0019915968,0.001932364,0.005781249],"category_scores_gemma":[0.027050672,0.0004952618,0.0006648727,0.001700822,0.0002550295,0.0036494927,0.0018392238,0.0019540212,0.0041206814],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002397458,0.0022452744,0.060153738,0.0006327002,0.00043245102,0.00036520712,0.0020687887,0.033507727,0.014506774,0.00840697,0.06715013,0.80813277],"study_design_scores_gemma":[0.00018793307,0.0009841272,0.012708109,0.00010802323,0.00024917873,0.00021768748,0.0008406983,0.9422513,0.010925959,0.0110966675,0.020287553,0.00014282136],"about_ca_topic_score_codex":0.018536326,"about_ca_topic_score_gemma":0.048473213,"teacher_disagreement_score":0.018536326,"about_ca_system_score_codex":0.0009011672,"about_ca_system_score_gemma":0.0033194688,"threshold_uncertainty_score":0.03685689},"labels":[],"label_agreement":null},{"id":"W4402237118","doi":"10.1007/s10664-024-10527-8","title":"An empirical study of token-based micro commits","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Radiation Effects in Electronics","field":"Engineering","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada","keywords":"Security token; Empirical research; Computer science; Mathematics; Statistics; Computer security","score_opus":0.011498103106017855,"score_gpt":0.28642940074989637,"score_spread":0.2749312976438785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402237118","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.993404,0.00007361849,0.0022497843,0.00029183258,0.000017395043,0.000048259742,0.000211311,0.000049709968,0.0036541275],"genre_scores_gemma":[0.9980817,0.000026058071,0.0004931823,0.00003113079,0.000008604398,0.000018168303,0.00013330317,0.000016561433,0.0011913765],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9919925,0.0040872805,0.00061223115,0.0007710308,0.0019941349,0.0005428829],"domain_scores_gemma":[0.65615314,0.2372056,0.05789785,0.023115786,0.019112313,0.006515369],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007722106,0.00021694165,0.00032823175,0.0013319275,0.0011301737,0.0022469724,0.0018161001,0.0010639182,0.011367945],"category_scores_gemma":[0.17787452,0.0003584197,0.00022271705,0.0025585725,0.0015246056,0.0036744324,0.0015936056,0.0027194251,0.0017732396],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014064166,0.002536468,0.89213806,0.0002071043,0.00012879366,0.000793968,0.008904027,0.006297063,0.0015422474,0.021777032,0.004021305,0.060247444],"study_design_scores_gemma":[0.00033971036,0.003188298,0.81732136,0.00023609409,0.00026482268,0.0016936443,0.03267209,0.085328124,0.0064145606,0.03590013,0.016452603,0.00018856661],"about_ca_topic_score_codex":0.0050139227,"about_ca_topic_score_gemma":0.005393841,"teacher_disagreement_score":0.011367945,"about_ca_system_score_codex":0.0011450807,"about_ca_system_score_gemma":0.0017968033,"threshold_uncertainty_score":0.040838897},"labels":[],"label_agreement":null},{"id":"W4402460553","doi":"10.1007/s10664-024-10536-7","title":"Quality issues in machine learning software systems","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Software quality; Quality (philosophy); Software engineering; Software; Software development; Operating system","score_opus":0.04127975929546383,"score_gpt":0.3288721992370755,"score_spread":0.28759243994161166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402460553","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24323438,0.02784397,0.6456067,0.04937709,0.0004587859,0.00017780505,0.00024828626,0.00038773945,0.032665312],"genre_scores_gemma":[0.96495473,0.0021875943,0.029771134,0.00041943925,0.00042790227,0.000065653614,0.00009093014,0.00009654995,0.001985966],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9849813,0.0072082453,0.0010048776,0.0010192348,0.005212692,0.0005737111],"domain_scores_gemma":[0.74204427,0.22442178,0.010624214,0.010812238,0.01083905,0.0012584548],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.022119662,0.0003836676,0.00086658815,0.0029478904,0.001267877,0.0047145886,0.001751217,0.002388411,0.0039873854],"category_scores_gemma":[0.17948304,0.0008601483,0.00067936716,0.0030671486,0.007053782,0.012757698,0.0019295636,0.0028904933,0.00022375275],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000051077153,0.00010983,0.00835857,0.000312152,0.000055290388,0.00007753504,0.00094465166,0.022344887,0.00023879764,0.922821,0.0016531812,0.043033086],"study_design_scores_gemma":[0.000016393902,0.000030214342,0.0024963934,0.00007917203,0.000015384909,0.000044658118,0.0002083434,0.032634877,0.00027435762,0.96202576,0.0021623077,0.000012156035],"about_ca_topic_score_codex":0.0036777505,"about_ca_topic_score_gemma":0.0018769678,"teacher_disagreement_score":0.97788036,"about_ca_system_score_codex":0.003705777,"about_ca_system_score_gemma":0.0020207143,"threshold_uncertainty_score":0.11698133},"labels":[],"label_agreement":null},{"id":"W4402810568","doi":"10.1007/s10664-024-10540-x","title":"An empirical study on developers’ shared conversations with ChatGPT in GitHub pull requests and issues","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kingston Health Sciences Centre; Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Empirical research; World Wide Web","score_opus":0.022466646346923743,"score_gpt":0.3286231495348124,"score_spread":0.3061565031878887,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402810568","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9964309,0.00005750894,0.0010666458,0.00023883602,0.000016318847,0.000092640475,0.00012360683,0.00007567047,0.0018979291],"genre_scores_gemma":[0.99572587,0.00007200631,0.001585345,0.00019316272,0.000026754367,0.00018844931,0.00030423354,0.000104552375,0.0017997338],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9821352,0.010180178,0.0010154652,0.0015799,0.0037763056,0.0013130298],"domain_scores_gemma":[0.68317676,0.24686833,0.025381943,0.010872542,0.02309958,0.010600816],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.012991813,0.0006965402,0.0005079514,0.0040327976,0.0037693686,0.003892378,0.0016265234,0.0025024095,0.0025089418],"category_scores_gemma":[0.13997583,0.00085225684,0.00031861622,0.0031820284,0.002551586,0.0045983577,0.005423313,0.0037686387,0.000941303],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005828495,0.0023335277,0.35272315,0.00049744145,0.00008020607,0.0026505592,0.586355,0.00040938327,0.008148996,0.0011006255,0.0026963823,0.042422015],"study_design_scores_gemma":[0.00009921704,0.0012425701,0.5169258,0.00044744535,0.00011259555,0.0013855401,0.4573595,0.0041138264,0.005138655,0.0011565012,0.011851235,0.0001671278],"about_ca_topic_score_codex":0.008414458,"about_ca_topic_score_gemma":0.01874821,"teacher_disagreement_score":0.9870082,"about_ca_system_score_codex":0.002620476,"about_ca_system_score_gemma":0.00364496,"threshold_uncertainty_score":0.06870812},"labels":[],"label_agreement":null},{"id":"W4402931849","doi":"10.1007/s10664-024-10538-5","title":"Examining ownership models in software teams","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Business; Software; Computer science; Process management; Software engineering; Knowledge management; Operating system","score_opus":0.05269670071759766,"score_gpt":0.28812360806570175,"score_spread":0.2354269073481041,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402931849","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9929389,0.00009831242,0.0033420515,0.00038195518,0.0000050735944,0.000014954291,0.000017967273,0.000006873331,0.003193888],"genre_scores_gemma":[0.99927384,0.000027574453,0.00032344772,0.000009674422,0.0000022011177,0.00000852485,0.000012905756,0.0000022366567,0.0003395857],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99572057,0.0028787954,0.0001484859,0.0002934763,0.0004668893,0.00049181224],"domain_scores_gemma":[0.8817297,0.09631769,0.012156153,0.0035266562,0.0038480498,0.002421729],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.00891143,0.00032063772,0.00034771024,0.0010900041,0.0011648629,0.0036672943,0.0013193978,0.0011505209,0.004846679],"category_scores_gemma":[0.063497104,0.0003086466,0.00043670725,0.0013940836,0.0015913934,0.0057394067,0.0019941174,0.0013552745,0.00026818045],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052214717,0.001225508,0.8009133,0.00008982933,0.00021661351,0.00025587704,0.009147295,0.036192507,0.0004858546,0.11025902,0.0014345949,0.039257485],"study_design_scores_gemma":[0.00020771282,0.0010606165,0.310522,0.00031026424,0.00039739796,0.00028328176,0.064085074,0.44292995,0.0016272723,0.17427218,0.0042196964,0.00008460825],"about_ca_topic_score_codex":0.016209038,"about_ca_topic_score_gemma":0.016491368,"teacher_disagreement_score":0.99108857,"about_ca_system_score_codex":0.0031226922,"about_ca_system_score_gemma":0.002431018,"threshold_uncertainty_score":0.047128737},"labels":[],"label_agreement":null},{"id":"W4403036283","doi":"10.1007/s10664-024-10548-3","title":"An empirical study on the effectiveness of large language models for SATD identification and classification","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba; Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Identification (biology); Computer science; Empirical research; Natural language processing; Artificial intelligence; Data science; Statistics; Mathematics; Biology","score_opus":0.027547982238886545,"score_gpt":0.3451586090035,"score_spread":0.31761062676461344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403036283","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94217074,0.0013393159,0.04735081,0.001391003,0.000091315516,0.00020172917,0.0018096904,0.0012010378,0.004444287],"genre_scores_gemma":[0.9724717,0.00023063301,0.023082172,0.00019752934,0.00004232367,0.000085571046,0.0028338467,0.00012588617,0.00093048875],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9847825,0.011712421,0.0007426565,0.0014367873,0.001056684,0.0002690011],"domain_scores_gemma":[0.6920501,0.2898922,0.0029988163,0.010154541,0.0038451988,0.0010591106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015189369,0.0011013073,0.00081594894,0.0022314962,0.00113818,0.0024639713,0.0020086744,0.0018480892,0.0033711274],"category_scores_gemma":[0.11443673,0.0004722135,0.0009694789,0.002179855,0.001260876,0.0063402974,0.00175773,0.0026596729,0.0016387127],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0060349857,0.0069002258,0.2963994,0.0011027524,0.0009099252,0.0005962156,0.0037512963,0.09666128,0.0071284347,0.011383045,0.017515473,0.5516169],"study_design_scores_gemma":[0.00034343515,0.0011339091,0.03934,0.00014393391,0.0003854061,0.0007106051,0.002456486,0.9291992,0.007151946,0.014302467,0.0047350796,0.00009753853],"about_ca_topic_score_codex":0.0062788273,"about_ca_topic_score_gemma":0.006135588,"teacher_disagreement_score":0.015189369,"about_ca_system_score_codex":0.0014939806,"about_ca_system_score_gemma":0.0013677562,"threshold_uncertainty_score":0.080330014},"labels":[],"label_agreement":null},{"id":"W4403255280","doi":"10.1007/s10664-024-10441-z","title":"Evaluating few-shot and contrastive learning methods for code clone detection","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Okanagan University College; University of British Columbia, Okanagan Campus; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; clone (Java method); Artificial intelligence; Code (set theory); Shot (pellet); Programming language; Biology; DNA; Genetics","score_opus":0.09238815332234587,"score_gpt":0.4441704756515731,"score_spread":0.35178232232922724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403255280","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6102431,0.021572413,0.33602673,0.0028277491,0.0011172328,0.000708886,0.0019013962,0.01667543,0.008927023],"genre_scores_gemma":[0.8476381,0.0011025143,0.13945791,0.0012943097,0.0002856198,0.00027300522,0.004502909,0.0006320029,0.0048135994],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99585366,0.0013684031,0.0001940167,0.001503742,0.00080778723,0.00027252876],"domain_scores_gemma":[0.9846585,0.011092508,0.0008159975,0.0012711341,0.0013924132,0.00076949474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007512184,0.0027867137,0.0019131626,0.0022082124,0.0008458235,0.0019495134,0.004643397,0.004193611,0.0015050604],"category_scores_gemma":[0.023489539,0.0006978136,0.0013792063,0.0010675906,0.0019175475,0.0043035233,0.0026778856,0.0044213724,0.0008420239],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019957516,0.0022085437,0.014532705,0.0011966467,0.00076595903,0.00025764472,0.00035844988,0.5333946,0.010001791,0.00373464,0.0135085685,0.41804475],"study_design_scores_gemma":[0.00004312371,0.00029439986,0.0009144303,0.00004008291,0.00003972303,0.000057326164,0.000038808063,0.9929969,0.0033753563,0.0015145045,0.000666724,0.000018792573],"about_ca_topic_score_codex":0.009342107,"about_ca_topic_score_gemma":0.010197704,"teacher_disagreement_score":0.009342107,"about_ca_system_score_codex":0.0031022925,"about_ca_system_score_gemma":0.001626989,"threshold_uncertainty_score":0.0397287},"labels":[],"label_agreement":null},{"id":"W4403359935","doi":"10.1007/s10664-024-10547-4","title":"Extracting microservices from monolithic systems using deep reinforcement learning","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Microservices; Reinforcement learning; Computer science; Artificial intelligence; Operating system","score_opus":0.021718347657193302,"score_gpt":0.271314858023251,"score_spread":0.2495965103660577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403359935","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34476885,0.00079556624,0.6452328,0.00038201,0.000082164646,0.000085056585,0.000709418,0.005962046,0.0019820402],"genre_scores_gemma":[0.9307462,0.00012157162,0.06636334,0.00004854665,0.000025426161,0.00003658425,0.00084473623,0.00015135042,0.0016623456],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997131,0.00003928141,0.000017639906,0.00010241517,0.00007072954,0.000056824272],"domain_scores_gemma":[0.99823284,0.00096082216,0.00021944045,0.0002728263,0.00022277309,0.000091259164],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00040877447,0.0009442529,0.0006596699,0.0011928901,0.00027429275,0.0005938267,0.0010201358,0.000942492,0.0013046458],"category_scores_gemma":[0.0027349358,0.00064191787,0.00081335724,0.0007625111,0.00045068553,0.0013052754,0.00078665407,0.0014378313,0.0006990167],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029306303,0.00029173808,0.015307292,0.00019078562,0.00011358277,0.00048217486,0.0001098142,0.6589956,0.019856468,0.004491067,0.0048171734,0.2950512],"study_design_scores_gemma":[0.0000045332326,0.000018757893,0.00065683527,0.000004297557,0.00000717641,0.000017336473,0.00000840609,0.99460924,0.0015747105,0.0028985408,0.0001967263,0.0000034169266],"about_ca_topic_score_codex":0.0051512932,"about_ca_topic_score_gemma":0.008577189,"teacher_disagreement_score":0.0051512932,"about_ca_system_score_codex":0.00064348814,"about_ca_system_score_gemma":0.0009400084,"threshold_uncertainty_score":0.010242641},"labels":[],"label_agreement":null},{"id":"W4403539235","doi":"10.1007/s10664-024-10570-5","title":"Bridging the language gap: an empirical study of bindings for open source machine learning libraries across software package ecosystems","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Bridging (networking); Computer science; Open source software; Open source; Empirical research; Software engineering; Software; Programming language","score_opus":0.03953274196089684,"score_gpt":0.34172572270005686,"score_spread":0.30219298073916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403539235","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99099064,0.0001073221,0.002613867,0.00049615913,0.0000065589634,0.000020433477,0.00005483139,0.00007556528,0.005634565],"genre_scores_gemma":[0.9968849,0.00004410376,0.0012997296,0.00015926115,0.0000067557494,0.000021521839,0.00008382611,0.00008816029,0.0014118438],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98867756,0.0057535004,0.00071128993,0.000991903,0.002715489,0.001150269],"domain_scores_gemma":[0.81971127,0.1225405,0.023770226,0.012045997,0.016481733,0.0054502715],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.013747537,0.00020640437,0.00043181443,0.002911683,0.003604331,0.0047553806,0.0020429292,0.001853136,0.00652807],"category_scores_gemma":[0.15339105,0.0005185639,0.0003015993,0.004574368,0.0053493087,0.015088783,0.0088381395,0.0032614213,0.0012894161],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010960987,0.002438248,0.6631146,0.0003778761,0.00009886343,0.0008640021,0.16301115,0.0014129771,0.006679662,0.046857078,0.0036470655,0.11040237],"study_design_scores_gemma":[0.0001404381,0.0011489317,0.6161222,0.00070891104,0.00023654132,0.0015530214,0.2644314,0.019641196,0.007795387,0.05528515,0.03271612,0.00022069871],"about_ca_topic_score_codex":0.011126456,"about_ca_topic_score_gemma":0.012072071,"teacher_disagreement_score":0.99795705,"about_ca_system_score_codex":0.0018670629,"about_ca_system_score_gemma":0.0036788075,"threshold_uncertainty_score":0.07270485},"labels":[],"label_agreement":null},{"id":"W4403547350","doi":"10.1007/s10664-024-10566-1","title":"Correction to: Examining ownership models in software teams","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Software; Data science; Software engineering; Programming language","score_opus":0.037156770907130494,"score_gpt":0.28165430634688066,"score_spread":0.24449753543975017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403547350","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006611904,0.0004017768,0.0008785196,0.06393403,0.9237718,0.00006485558,0.007331071,0.00078458904,0.0021721579],"genre_scores_gemma":[0.106790535,0.0048508258,0.0117072025,0.10099044,0.35789534,0.0017661358,0.018648801,0.005288313,0.3920624],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98738223,0.001525764,0.002788946,0.002308064,0.004520491,0.0014745852],"domain_scores_gemma":[0.7508577,0.059896883,0.010048557,0.02378184,0.14711721,0.00829786],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0071286373,0.002845309,0.0053975256,0.009088733,0.007754198,0.007893847,0.008239678,0.013683814,0.20278734],"category_scores_gemma":[0.2502268,0.0023216754,0.0027184102,0.011685997,0.004667129,0.006418139,0.005272781,0.014029354,0.07620586],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044253244,0.000010464228,0.00027612684,0.00012893611,0.000025342095,0.0001696596,0.000063376545,0.00006998558,0.00002560429,0.0011561842,0.9949986,0.0030315048],"study_design_scores_gemma":[0.0005026019,0.000089448265,0.012867721,0.001454893,0.00025227648,0.0014319777,0.0015270417,0.0030502556,0.0010482379,0.012212396,0.9651837,0.0003795438],"about_ca_topic_score_codex":0.04497355,"about_ca_topic_score_gemma":0.045437023,"teacher_disagreement_score":0.20278734,"about_ca_system_score_codex":0.009118656,"about_ca_system_score_gemma":0.01164853,"threshold_uncertainty_score":0.6783912},"labels":[],"label_agreement":null},{"id":"W4403651448","doi":"10.1007/s10664-024-10568-z","title":"The downside of functional constructs: a quantitative and qualitative analysis of their fix-inducing effects","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Ministero dell'Università e della Ricerca","keywords":"Qualitative analysis; Quantitative analysis (chemistry); Downside risk; Computer science; Qualitative research; Chemistry; Economics; Sociology; Chromatography; Financial economics","score_opus":0.03409354701278354,"score_gpt":0.3381684211001117,"score_spread":0.30407487408732814,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403651448","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9411356,0.0003783834,0.04537841,0.00035472008,0.000020868205,0.0001424596,0.00020096578,0.00017426522,0.012214313],"genre_scores_gemma":[0.98708296,0.000116384996,0.011805455,0.00005348787,0.000007737994,0.000085710664,0.00008478209,0.00009898448,0.00066445593],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.99496645,0.0022391044,0.0002775899,0.0004392447,0.0018227493,0.00025480028],"domain_scores_gemma":[0.8143054,0.16177931,0.0061880173,0.009640874,0.007254903,0.0008314608],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008506766,0.00034263643,0.00031012818,0.002076471,0.0008498771,0.0009949224,0.00054264534,0.0005803342,0.0049135825],"category_scores_gemma":[0.06689005,0.0002952714,0.00043015214,0.0010638812,0.0019429862,0.0019041001,0.0013047602,0.0015700803,0.00021344953],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038209509,0.0012728883,0.17209977,0.0035617894,0.0003273404,0.0009745631,0.02177597,0.010141592,0.28158695,0.13603576,0.0018068233,0.36659554],"study_design_scores_gemma":[0.00022847038,0.004021725,0.63032883,0.0009570028,0.0012681886,0.0022545746,0.01652616,0.026363581,0.22937003,0.06910177,0.019364072,0.00021569383],"about_ca_topic_score_codex":0.00061764993,"about_ca_topic_score_gemma":0.0009398369,"teacher_disagreement_score":0.008506766,"about_ca_system_score_codex":0.0007392488,"about_ca_system_score_gemma":0.0007693853,"threshold_uncertainty_score":0.044988573},"labels":[],"label_agreement":null},{"id":"W4403887388","doi":"10.1007/s10664-024-10549-2","title":"Towards effectively testing machine translation systems from white-box perspectives","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; University of Waterloo","funders":"","keywords":"Computer science; Translation (biology); Machine translation; White box; Systems engineering; Artificial intelligence; Engineering; Software engineering","score_opus":0.02217214222265518,"score_gpt":0.27888827832438084,"score_spread":0.25671613610172567,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403887388","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09494362,0.00028431445,0.8935737,0.001760985,0.000084495725,0.00018547174,0.00021766385,0.0052251434,0.0037246454],"genre_scores_gemma":[0.5383083,0.00015737486,0.457419,0.0006555539,0.00009743655,0.00019550534,0.00054820866,0.0010893457,0.0015293489],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.96979386,0.021093413,0.0013506467,0.002358413,0.00448338,0.0009203235],"domain_scores_gemma":[0.83437306,0.1376906,0.003751718,0.014566704,0.008431872,0.0011860877],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017273396,0.0017123667,0.0012218999,0.0023808877,0.0010312756,0.005320158,0.0030245604,0.0037891674,0.005006432],"category_scores_gemma":[0.1130148,0.0012648002,0.0009909312,0.0014019621,0.0037120376,0.013016649,0.003983439,0.0034023423,0.0012394241],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013670146,0.002335174,0.022116413,0.0018097475,0.00048739035,0.0021664202,0.004273767,0.12451614,0.098821335,0.2421831,0.010367657,0.48955595],"study_design_scores_gemma":[0.00017233477,0.00041575776,0.001874954,0.00020380654,0.000119483404,0.00044722244,0.0009892301,0.68732035,0.055132903,0.24790622,0.0053597414,0.000058061098],"about_ca_topic_score_codex":0.001810842,"about_ca_topic_score_gemma":0.0028131332,"teacher_disagreement_score":0.017273396,"about_ca_system_score_codex":0.001103138,"about_ca_system_score_gemma":0.0033094187,"threshold_uncertainty_score":0.09135157},"labels":[],"label_agreement":null},{"id":"W4403946944","doi":"10.1007/s10664-024-10560-7","title":"A qualitative study on refactorings induced by code review","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Code (set theory); Programming language; Qualitative research; Software engineering; Sociology","score_opus":0.0852812258416174,"score_gpt":0.41490363149144455,"score_spread":0.32962240564982714,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403946944","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95841634,0.0006811531,0.015602919,0.0045101596,0.00023100372,0.0009860046,0.00070021994,0.00016488055,0.018707262],"genre_scores_gemma":[0.98478216,0.00050580705,0.005477158,0.0017988152,0.00006060658,0.0008451192,0.0003004193,0.00014612853,0.0060838074],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9702351,0.021244284,0.0010053546,0.0015125421,0.0042226813,0.0017801234],"domain_scores_gemma":[0.68343306,0.27263707,0.010899252,0.0059666703,0.022233607,0.0048303464],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.020882344,0.0005430647,0.0005120643,0.0030966098,0.0077236993,0.0032719232,0.0019778202,0.0026454711,0.0043702777],"category_scores_gemma":[0.105523,0.00060381956,0.00039893197,0.0023654127,0.004020767,0.003344868,0.0041888044,0.0033336736,0.00071701535],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001463651,0.00029088336,0.010139069,0.0009338329,0.000015389827,0.0026550405,0.9474417,0.00012682051,0.009426288,0.002809386,0.002477184,0.023538016],"study_design_scores_gemma":[0.00004222695,0.00037005052,0.015841687,0.00092673744,0.000031488627,0.0013063436,0.9332808,0.00063039566,0.006296795,0.0018573321,0.039333086,0.000083127634],"about_ca_topic_score_codex":0.0054790406,"about_ca_topic_score_gemma":0.013908016,"teacher_disagreement_score":0.97911763,"about_ca_system_score_codex":0.0058104545,"about_ca_system_score_gemma":0.00716713,"threshold_uncertainty_score":0.11043769},"labels":[],"label_agreement":null},{"id":"W4404199042","doi":"10.1007/s10664-024-10579-w","title":"Towards enhancing the reproducibility of deep learning bugs: an empirical study","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Dalhousie University","funders":"","keywords":"Reproducibility; Empirical research; Computer science; Data science; Artificial intelligence; Statistics; Mathematics","score_opus":0.02200891701765622,"score_gpt":0.32076736995582494,"score_spread":0.2987584529381687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404199042","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8724037,0.001311158,0.11813426,0.0014164897,0.00012540777,0.00017383767,0.00051439885,0.0017735214,0.0041471166],"genre_scores_gemma":[0.98532605,0.000098234406,0.013450378,0.00011342048,0.000030840158,0.000037005615,0.00024133707,0.00017604223,0.0005265636],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9721835,0.013411736,0.0019755305,0.0043612015,0.007370084,0.0006979575],"domain_scores_gemma":[0.40540993,0.44339937,0.031094229,0.09931839,0.018890902,0.0018872397],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.028014367,0.0007686686,0.00069220946,0.0016683978,0.0009113282,0.0018351148,0.0025473335,0.0021657867,0.0026685847],"category_scores_gemma":[0.35097265,0.0005191975,0.000932158,0.0012529722,0.003048871,0.0043176673,0.002505649,0.0034795334,0.00055023807],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033301055,0.0036839,0.32348263,0.0017120847,0.0012068333,0.0010893524,0.0032805805,0.22553591,0.019356305,0.043540873,0.012307475,0.36147392],"study_design_scores_gemma":[0.00042356562,0.0027512307,0.06616565,0.00049107795,0.00053313834,0.0020046826,0.0010210312,0.7969645,0.024881748,0.09724044,0.0073593375,0.00016361711],"about_ca_topic_score_codex":0.001399817,"about_ca_topic_score_gemma":0.0014084888,"teacher_disagreement_score":0.97198564,"about_ca_system_score_codex":0.0011796593,"about_ca_system_score_gemma":0.0017354623,"threshold_uncertainty_score":0.14815587},"labels":[],"label_agreement":null},{"id":"W4404406203","doi":"10.1007/s10664-024-10564-3","title":"Can search-based testing with pareto optimization effectively cover failure-revealing test inputs?","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; HORIZON EUROPE Reforming and enhancing the European Research and Innovation system; Technische Universität München; European Commission","keywords":"Cover (algebra); Pareto principle; Reliability engineering; Computer science; Engineering; Mathematical optimization; Mathematics; Mechanical engineering","score_opus":0.02105501579744261,"score_gpt":0.25535485620899456,"score_spread":0.23429984041155194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404406203","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42423457,0.00052705785,0.56591296,0.00087128795,0.000056495355,0.000153588,0.000114438946,0.00079714163,0.0073324023],"genre_scores_gemma":[0.9578573,0.00006134636,0.0413006,0.0001070322,0.000008509154,0.00006598467,0.00007140204,0.000048936927,0.00047873225],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983284,0.00073300896,0.00007184039,0.00016808041,0.00043126033,0.000267376],"domain_scores_gemma":[0.9927468,0.005231422,0.00046997963,0.0006399328,0.0006762416,0.00023573873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031034062,0.00093683606,0.0008325371,0.0010681466,0.0004025681,0.00086304685,0.001134423,0.0009469823,0.0016325449],"category_scores_gemma":[0.0141244,0.00031523878,0.00083703536,0.0006115877,0.0011913315,0.0013600314,0.00095183484,0.0009317509,0.00023417424],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000105104664,0.00010228896,0.0029343492,0.00004968298,0.00004001261,0.00006973785,0.000040958275,0.9617073,0.002202952,0.0041609425,0.0003804757,0.028206129],"study_design_scores_gemma":[0.000013902201,0.00008284969,0.00042742642,0.000012445486,0.00000946207,0.000016859985,0.000024842655,0.9940865,0.0012588763,0.0038742945,0.00018896026,0.0000034986965],"about_ca_topic_score_codex":0.005190806,"about_ca_topic_score_gemma":0.0037244658,"teacher_disagreement_score":0.005190806,"about_ca_system_score_codex":0.0011302092,"about_ca_system_score_gemma":0.0016321199,"threshold_uncertainty_score":0.016412556},"labels":[],"label_agreement":null},{"id":"W4404860113","doi":"10.1007/s10664-024-10589-8","title":"Contrasting test selection, prioritization, and batch testing at scale","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; Concordia University","keywords":"Prioritization; Selection (genetic algorithm); Scale (ratio); Test (biology); Computer science; Regression testing; Reliability engineering; Engineering; Machine learning; Biology; Geography; Management science; Cartography; Operating system; Software","score_opus":0.016660816534096935,"score_gpt":0.2537672231562379,"score_spread":0.23710640662214097,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404860113","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8943175,0.0019294876,0.08601288,0.0032795845,0.00012295633,0.00026586722,0.00022239644,0.00064694486,0.013202383],"genre_scores_gemma":[0.98673904,0.00007881837,0.012021151,0.0001860204,0.000047976024,0.00006975466,0.00008790527,0.00006861456,0.00070072734],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.96169186,0.02681852,0.0010594625,0.0031087443,0.0055371625,0.0017843539],"domain_scores_gemma":[0.45321968,0.49802822,0.015919702,0.0189081,0.010157303,0.0037669253],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.046024665,0.0012004359,0.0014860422,0.0018941776,0.0010430013,0.0030197862,0.003107627,0.0020464475,0.0038235756],"category_scores_gemma":[0.28283623,0.00060154434,0.000863053,0.0022393207,0.0040273466,0.0070025884,0.0023197555,0.0026111295,0.00036534842],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013953913,0.0041357824,0.28279185,0.0011102263,0.0016324478,0.00083405577,0.0034516463,0.1524799,0.013311325,0.111899495,0.007232205,0.40716714],"study_design_scores_gemma":[0.0023917828,0.0069831647,0.32130638,0.00028773525,0.0011905495,0.0005499144,0.0031692986,0.3992859,0.00986393,0.25047982,0.004226229,0.000265319],"about_ca_topic_score_codex":0.011993962,"about_ca_topic_score_gemma":0.013060074,"teacher_disagreement_score":0.046024665,"about_ca_system_score_codex":0.0029649367,"about_ca_system_score_gemma":0.0035841744,"threshold_uncertainty_score":0.24340463},"labels":[],"label_agreement":null},{"id":"W4405259287","doi":"10.1007/s10664-024-10597-8","title":"Harnessing pre-trained generalist agents for software engineering tasks","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Artificial intelligence; Reinforcement learning; Software engineering; Generalizability theory; Scheduling (production processes); Machine learning; Software; Domain (mathematical analysis); Human–computer interaction; Engineering; Operations management","score_opus":0.03556022016790315,"score_gpt":0.3148634202315637,"score_spread":0.27930320006366055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405259287","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.62802964,0.0003782523,0.32762682,0.0007724147,0.00019025734,0.00051031227,0.00018534489,0.0057170033,0.036589894],"genre_scores_gemma":[0.8741548,0.00019760775,0.1099425,0.0004115148,0.00004997045,0.00018795808,0.0003745756,0.00026055184,0.014420534],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993656,0.00019909507,0.000030487645,0.00019339516,0.00013585186,0.00007560098],"domain_scores_gemma":[0.99622923,0.0014999069,0.0002870033,0.0009911491,0.0006378599,0.00035477342],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013008745,0.0006634757,0.00044571763,0.00045406283,0.0004292776,0.0011677525,0.0012311325,0.0010687689,0.0060692984],"category_scores_gemma":[0.007751285,0.00046216237,0.00029245156,0.00038313726,0.00048056876,0.0013012084,0.0020978525,0.0014273361,0.0028948314],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011413154,0.002910708,0.03333971,0.00054472656,0.00024353249,0.000576374,0.0032179414,0.076391645,0.1258015,0.0066806544,0.009739084,0.7394127],"study_design_scores_gemma":[0.00031067137,0.0024782894,0.029582808,0.00016897355,0.00031075155,0.00062811654,0.0020905721,0.8290638,0.06173095,0.020788083,0.05270581,0.00014122254],"about_ca_topic_score_codex":0.0021513454,"about_ca_topic_score_gemma":0.0059024733,"teacher_disagreement_score":0.0060692984,"about_ca_system_score_codex":0.00047066194,"about_ca_system_score_gemma":0.0011421628,"threshold_uncertainty_score":0.020303845},"labels":[],"label_agreement":null},{"id":"W4406172285","doi":"10.1007/s10664-024-10555-4","title":"Lightweight dynamic build batching algorithms for continuous integration","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Process (computing); Batch processing; Quality (philosophy); Software; Quality assurance; Distributed computing; Industrial engineering; Real-time computing; Software engineering; Engineering; Operating system; Operations management","score_opus":0.014474937432760255,"score_gpt":0.3036872083430948,"score_spread":0.2892122709103346,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406172285","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011334809,0.00022669432,0.9707445,0.00013401598,0.00010274383,0.000112100635,0.00017516728,0.01585379,0.001316185],"genre_scores_gemma":[0.1656081,0.00012816589,0.82667315,0.00014364648,0.0001001948,0.00025000647,0.000911056,0.0024804163,0.003705243],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9956683,0.0006410976,0.00039662942,0.0009448172,0.0018368,0.0005123484],"domain_scores_gemma":[0.98669785,0.005249305,0.00074714684,0.0054043117,0.0013015588,0.0005998689],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029450704,0.0018815473,0.0019467174,0.0016594945,0.0013698693,0.002933126,0.0054858443,0.0017725712,0.015391985],"category_scores_gemma":[0.015835615,0.0018650885,0.0016568595,0.002308937,0.0015134924,0.0050861067,0.006259702,0.0042699957,0.005872539],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013890686,0.0003866611,0.0032499137,0.00026104253,0.0001469844,0.00020532458,0.00041722157,0.07761628,0.027258448,0.019783836,0.016591918,0.85269326],"study_design_scores_gemma":[0.00020864712,0.00015741478,0.0008767395,0.000030399011,0.00006815738,0.00013896496,0.00011641048,0.94441634,0.014277521,0.034064747,0.0055946936,0.00004994825],"about_ca_topic_score_codex":0.0063271946,"about_ca_topic_score_gemma":0.0117609445,"teacher_disagreement_score":0.015391985,"about_ca_system_score_codex":0.0013568883,"about_ca_system_score_gemma":0.0029560206,"threshold_uncertainty_score":0.05149126},"labels":[],"label_agreement":null},{"id":"W4406405723","doi":"10.1007/s10664-024-10611-z","title":"Negativity in self-admitted technical debt: how sentiment influences prioritization","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Auditing, Earnings Management, Governance","field":"Business, Management and Accounting","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Negativity effect; Prioritization; Debt; Computer science; Psychology; Cognitive psychology; Business; Process management; Finance","score_opus":0.007187406690226814,"score_gpt":0.2287443263012959,"score_spread":0.22155691961106908,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406405723","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9953205,0.00005157418,0.00076985924,0.00027405695,0.00001459384,0.000017884568,0.00003405073,0.000011679226,0.0035057552],"genre_scores_gemma":[0.9989778,0.000036774873,0.000500779,0.00011416758,0.000004776885,0.00001437375,0.000037415994,0.00000864922,0.0003053206],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99532026,0.003166785,0.00019235861,0.00033516757,0.0007502763,0.00023520309],"domain_scores_gemma":[0.9605047,0.025634812,0.007608603,0.00084295176,0.0035025855,0.0019064051],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004867002,0.00023336313,0.00020307077,0.00046908337,0.0007969953,0.0023550235,0.00030496198,0.00066236465,0.0020690025],"category_scores_gemma":[0.039104853,0.0002415656,0.00020684942,0.0003192535,0.0010044992,0.0010659432,0.0011180813,0.0010694014,0.0003060589],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000760956,0.0004805809,0.7863595,0.0005178062,0.00012573371,0.00083719526,0.14703651,0.0008602642,0.023551386,0.0021802078,0.0038985866,0.033391252],"study_design_scores_gemma":[0.00004484648,0.00039989172,0.8765641,0.0001751167,0.000090133406,0.00034425757,0.10135156,0.0053039477,0.003536557,0.0017778527,0.0102889165,0.000122772],"about_ca_topic_score_codex":0.002907133,"about_ca_topic_score_gemma":0.0038187888,"teacher_disagreement_score":0.004867002,"about_ca_system_score_codex":0.0012406328,"about_ca_system_score_gemma":0.00048754405,"threshold_uncertainty_score":0.025739491},"labels":[],"label_agreement":null},{"id":"W4407277384","doi":"10.1007/s10664-024-10609-7","title":"UPC sentinel: An accurate approach for detecting upgradeability proxy contracts in Ethereum","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Proxy (statistics); Computer science; Engineering; Machine learning","score_opus":0.03488539790364532,"score_gpt":0.326717984410252,"score_spread":0.2918325865066067,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407277384","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45634282,0.0010784997,0.4988465,0.00084359146,0.0002595048,0.000492163,0.008121933,0.020821372,0.013193676],"genre_scores_gemma":[0.82896894,0.00020170229,0.16118024,0.00024341267,0.000058049438,0.00015795344,0.0049654525,0.0005309244,0.003693363],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99132735,0.0015160501,0.0006259823,0.0010701393,0.0048485873,0.00061183673],"domain_scores_gemma":[0.97545683,0.009571331,0.004144902,0.0052281497,0.0048184,0.00078034785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068490854,0.00086470874,0.0007821155,0.006083718,0.0006833981,0.0019618436,0.001993849,0.002215171,0.0035052253],"category_scores_gemma":[0.033676255,0.00042874538,0.0004697494,0.0026019178,0.00063911953,0.003951066,0.0022965597,0.0017877596,0.0012257192],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014474195,0.00075638306,0.3178537,0.000662634,0.00031954265,0.0016375124,0.0009296052,0.06455966,0.047510244,0.043618828,0.034198567,0.48650596],"study_design_scores_gemma":[0.00011347872,0.00032127523,0.05063699,0.00016065914,0.00011355873,0.00096218084,0.0004933604,0.8448142,0.05025655,0.02795029,0.024051916,0.00012552722],"about_ca_topic_score_codex":0.005484115,"about_ca_topic_score_gemma":0.007028841,"teacher_disagreement_score":0.0068490854,"about_ca_system_score_codex":0.0010829062,"about_ca_system_score_gemma":0.0024403029,"threshold_uncertainty_score":0.036221802},"labels":[],"label_agreement":null},{"id":"W4407551092","doi":"10.1007/s10664-025-10614-4","title":"Bugs in large language models generated code: an empirical study","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de recherche du Québec; Natural Sciences and Engineering Research Council of Canada; Consortium de Recherche et d’innovation en Aérospatiale au Québec; Canadian Institute for Advanced Research","keywords":"Computer science; Programming language; Empirical research; Code (set theory); Statistics; Mathematics","score_opus":0.019862348010958434,"score_gpt":0.32677663977065224,"score_spread":0.3069142917596938,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407551092","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99858874,0.000052547046,0.00092747464,0.000036824975,0.00000265776,0.000022126498,0.00007989356,0.000050745337,0.00023900926],"genre_scores_gemma":[0.998489,0.000041507654,0.0008973554,0.000026243642,0.0000052614373,0.000023423976,0.00026912114,0.00003221236,0.00021580586],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99634725,0.0016318305,0.0002567779,0.00048205268,0.0011370684,0.00014506334],"domain_scores_gemma":[0.73847705,0.22292149,0.018138088,0.009292171,0.010072479,0.0010987278],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034137764,0.000452118,0.00023482059,0.0016370184,0.0005232767,0.0007353479,0.00097456784,0.0009871216,0.0014806943],"category_scores_gemma":[0.091930024,0.0003654459,0.0003807407,0.0010591601,0.0013586223,0.0016785637,0.0006738809,0.0013355013,0.0003576337],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018963201,0.009793608,0.84998655,0.0005033607,0.00021709032,0.0016123041,0.0060210754,0.020900479,0.010654825,0.002608618,0.003226934,0.092578806],"study_design_scores_gemma":[0.00051751337,0.007370322,0.7030223,0.00032753721,0.00035808832,0.006647916,0.0048996853,0.24763268,0.01977598,0.00504439,0.0042368257,0.00016675587],"about_ca_topic_score_codex":0.0022657984,"about_ca_topic_score_gemma":0.0030615956,"teacher_disagreement_score":0.0034137764,"about_ca_system_score_codex":0.0007042586,"about_ca_system_score_gemma":0.0005988066,"threshold_uncertainty_score":0.018054008},"labels":[],"label_agreement":null},{"id":"W4407986251","doi":"10.1007/s10664-025-10619-z","title":"An empirical study on LLM-based classification of requirements-related provisions in food-safety regulations","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Technology and Data Analysis","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Ontario Centre of Innovation","keywords":"Food safety; Empirical research; Business; Computer science; Food science; Chemistry; Statistics; Mathematics","score_opus":0.032420337496137465,"score_gpt":0.33388914271590625,"score_spread":0.30146880521976877,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407986251","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99812835,0.000039460865,0.0005404312,0.000060401242,0.00000388087,0.000029292572,0.0000969792,0.0000047403532,0.0010965032],"genre_scores_gemma":[0.9984786,0.000021931566,0.0007057786,0.000030510997,0.000005061603,0.000034166525,0.0003337463,0.0000050411772,0.00038512083],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.989749,0.0064778673,0.0007656871,0.0008712402,0.0015109213,0.00062520057],"domain_scores_gemma":[0.8430997,0.1209858,0.019515887,0.00479085,0.009512971,0.0020947307],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011935806,0.00034910362,0.00031679036,0.00290462,0.0009358088,0.0015525832,0.0014208752,0.00085309526,0.004554838],"category_scores_gemma":[0.069113664,0.00020972747,0.00073432474,0.0031032972,0.0011836394,0.0026600866,0.0016356964,0.0019051878,0.00076172495],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030082732,0.001319373,0.9791504,0.00005861076,0.000052324434,0.000089473324,0.0022933225,0.00051759544,0.00046355353,0.0010308988,0.0003784316,0.014345144],"study_design_scores_gemma":[0.000037173926,0.0005033804,0.97136575,0.00009007306,0.000093308554,0.00013755212,0.013262402,0.011384994,0.00090494566,0.00056970253,0.0016271292,0.000023427241],"about_ca_topic_score_codex":0.010968636,"about_ca_topic_score_gemma":0.011157563,"teacher_disagreement_score":0.011935806,"about_ca_system_score_codex":0.001954341,"about_ca_system_score_gemma":0.0015770766,"threshold_uncertainty_score":0.063123286},"labels":[],"label_agreement":null},{"id":"W4407986476","doi":"10.1007/s10664-025-10618-0","title":"Evaluating interactive documentation for programmers","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Documentation; Computer science; World Wide Web; Software engineering; Programming language","score_opus":0.04750618481790624,"score_gpt":0.41323817976843863,"score_spread":0.3657319949505324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407986476","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9755727,0.0005164751,0.014571881,0.00021075118,0.000045740475,0.00018781425,0.0001602158,0.0015912196,0.0071433657],"genre_scores_gemma":[0.97256136,0.00019045892,0.023962332,0.000061846105,0.000029138877,0.000105446554,0.0005075435,0.00025688164,0.002325118],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98577785,0.007269369,0.0010237555,0.00081282074,0.0046645245,0.00045162116],"domain_scores_gemma":[0.7851467,0.15683575,0.013870568,0.014107821,0.02555297,0.004486238],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010161582,0.0009799625,0.00084742083,0.0022913225,0.0011043303,0.0037444402,0.001555256,0.0019670967,0.004311485],"category_scores_gemma":[0.16134547,0.00047942656,0.00044915397,0.0016756401,0.00069476257,0.0029797165,0.0020344597,0.0012407133,0.0009837829],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008655415,0.007034823,0.12522174,0.0014286612,0.0004671304,0.00070318184,0.00846014,0.037291843,0.018828016,0.0032410948,0.006501003,0.7821669],"study_design_scores_gemma":[0.0034786016,0.036334567,0.2538049,0.0015982052,0.0019740395,0.0016991849,0.013093629,0.5502153,0.09596464,0.0146487495,0.026720934,0.00046730886],"about_ca_topic_score_codex":0.004453322,"about_ca_topic_score_gemma":0.0061037233,"teacher_disagreement_score":0.010161582,"about_ca_system_score_codex":0.0016472593,"about_ca_system_score_gemma":0.0024823935,"threshold_uncertainty_score":0.053740203},"labels":[],"label_agreement":null},{"id":"W4408054408","doi":"10.1007/s10664-024-10610-0","title":"Assessing the adoption of security policies by developers in terraform across different cloud providers","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Information and Cyber Security","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de recherche du Québec; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Cloud computing; Business; Computer security; Cloud service provider; Cloud computing security; Internet privacy; Computer science","score_opus":0.014692344055413297,"score_gpt":0.29999987601119277,"score_spread":0.28530753195577946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408054408","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99729913,0.00006348328,0.00080316816,0.0001255249,0.0000029894566,0.000057775076,0.00032114642,0.00015535625,0.0011713828],"genre_scores_gemma":[0.99142754,0.00012783914,0.0054494976,0.00006696316,0.000006382693,0.00013638355,0.0016414396,0.00014500297,0.0009990647],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9921485,0.0021470673,0.0009155039,0.0013284701,0.002766613,0.00069369713],"domain_scores_gemma":[0.8453686,0.076496415,0.03587568,0.011968438,0.025821438,0.0044693365],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015874969,0.00044979114,0.00026834465,0.0050541135,0.00088331517,0.001969507,0.0007364682,0.0007645593,0.00085532875],"category_scores_gemma":[0.087660104,0.0004915363,0.00034769086,0.004227854,0.0016879644,0.0029243492,0.0017813856,0.0011893897,0.00040665898],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015138477,0.00025718237,0.9328825,0.000189674,0.000050864768,0.0002380738,0.015518094,0.001372012,0.0035960348,0.00072647125,0.0013365145,0.043681137],"study_design_scores_gemma":[0.000018217175,0.0003475965,0.97592473,0.00014367534,0.00003928513,0.0002974888,0.009678844,0.0054756207,0.0020949144,0.0003919115,0.0055261585,0.00006160458],"about_ca_topic_score_codex":0.014650667,"about_ca_topic_score_gemma":0.021210391,"teacher_disagreement_score":0.015874969,"about_ca_system_score_codex":0.0019370579,"about_ca_system_score_gemma":0.0022863692,"threshold_uncertainty_score":0.083955824},"labels":[],"label_agreement":null},{"id":"W4408069583","doi":"10.1007/s10664-025-10616-2","title":"WIA-SZZ: Work item aware SZZ","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Work (physics); Engineering; Mechanical engineering","score_opus":0.0177152079228471,"score_gpt":0.2842851256222398,"score_spread":0.2665699176993927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408069583","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032600593,0.00058282336,0.33466172,0.00074152707,0.0004192902,0.0015785985,0.09026908,0.5134727,0.025673741],"genre_scores_gemma":[0.17383416,0.00058131467,0.5981326,0.00064536306,0.0001556779,0.0024710929,0.16390324,0.018196555,0.042079955],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982437,0.0002319233,0.00017969291,0.0004419606,0.0007207162,0.00018205504],"domain_scores_gemma":[0.9968045,0.0006631133,0.00024334999,0.0014168209,0.0006818901,0.00019034858],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001287796,0.0018974203,0.001076189,0.0031634879,0.0004992278,0.0021899617,0.0025461572,0.0008611985,0.028726475],"category_scores_gemma":[0.008261156,0.0010043273,0.0015085356,0.0021420838,0.00030420642,0.0035229044,0.004710234,0.0013883184,0.025143735],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023230405,0.00083840993,0.013849091,0.0011083082,0.00050981226,0.00019669368,0.00036292232,0.00480368,0.017973572,0.007623559,0.3149212,0.6354897],"study_design_scores_gemma":[0.0016364404,0.0009082046,0.04460111,0.00031885633,0.00075105735,0.00073606713,0.00068515923,0.3154838,0.10613168,0.04901392,0.47917873,0.00055497367],"about_ca_topic_score_codex":0.0028458268,"about_ca_topic_score_gemma":0.0062377728,"teacher_disagreement_score":0.028726475,"about_ca_system_score_codex":0.00042544494,"about_ca_system_score_gemma":0.0011361338,"threshold_uncertainty_score":0.096099675},"labels":[],"label_agreement":null},{"id":"W4408089088","doi":"10.1007/s10664-025-10634-0","title":"Fixer-level supervised contrastive learning for bug assignment","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Artificial intelligence; Natural language processing; Machine learning","score_opus":0.03540607240363587,"score_gpt":0.3021084142661999,"score_spread":0.26670234186256403,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408089088","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11997999,0.0011610042,0.8658953,0.0004143509,0.00016909579,0.00016531044,0.00072324136,0.008120787,0.0033708264],"genre_scores_gemma":[0.7564338,0.00015507427,0.23550095,0.00032640854,0.0001547782,0.00015613383,0.002598404,0.0005666358,0.0041077468],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99757904,0.00083586416,0.00011202119,0.0008624039,0.00042112079,0.0001895583],"domain_scores_gemma":[0.98897564,0.0070190323,0.00063052744,0.0016155108,0.0014482643,0.00031106337],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035625426,0.0011288207,0.0014197892,0.0020121366,0.0008094157,0.0011499039,0.0038392125,0.002226822,0.0033674287],"category_scores_gemma":[0.014094843,0.00044876433,0.0008362628,0.0011876198,0.00093292625,0.0024527677,0.0023738998,0.0032878781,0.0011872584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014241136,0.00097499933,0.012149215,0.00042437084,0.000314726,0.00016613229,0.00020014418,0.091114506,0.022100085,0.008111721,0.013056705,0.84996337],"study_design_scores_gemma":[0.00007083501,0.0002443716,0.0020267242,0.000023880099,0.00005382218,0.00009129818,0.000028914827,0.98169655,0.0060343584,0.008221787,0.0014873053,0.000020114105],"about_ca_topic_score_codex":0.0034988553,"about_ca_topic_score_gemma":0.00890047,"teacher_disagreement_score":0.0038392125,"about_ca_system_score_codex":0.0011540448,"about_ca_system_score_gemma":0.0016191461,"threshold_uncertainty_score":0.01884073},"labels":[],"label_agreement":null},{"id":"W4408204750","doi":"10.1007/s10664-025-10631-3","title":"Towards semantic versioning of open pre-trained language model releases on hugging face","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Face recognition and analysis","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); Queen's University","funders":"","keywords":"Software versioning; Computer science; Face (sociological concept); Natural language processing; Artificial intelligence; World Wide Web; Information retrieval; Linguistics; Programming language; Software","score_opus":0.01979562248878209,"score_gpt":0.3129141107030443,"score_spread":0.29311848821426223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408204750","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20806864,0.0005569404,0.6673707,0.0008211073,0.0008997053,0.00025184234,0.0025278814,0.11005265,0.009450513],"genre_scores_gemma":[0.69940406,0.00027581645,0.27646285,0.00035956304,0.00014649674,0.00017115987,0.008040638,0.008053001,0.0070864037],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9963684,0.0008540732,0.0002883057,0.0009559777,0.0011560521,0.0003772535],"domain_scores_gemma":[0.98410356,0.0036697446,0.0005492181,0.009406694,0.0019864596,0.00028420682],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.0037157005,0.0010611684,0.0008340672,0.0011930197,0.0007545041,0.0035552445,0.0025359783,0.0016302903,0.005018873],"category_scores_gemma":[0.024683237,0.0009668671,0.0015121901,0.00075592013,0.001200993,0.007616,0.0043750647,0.0033444436,0.0029874058],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015373429,0.00075324887,0.011740765,0.0004245183,0.00024642024,0.0009877908,0.0019142213,0.067647085,0.049922768,0.034415755,0.043874647,0.78653544],"study_design_scores_gemma":[0.00010498938,0.00028280172,0.003671792,0.00013075813,0.00012275064,0.00040347388,0.00048680473,0.854009,0.07072711,0.048902076,0.021032844,0.00012562973],"about_ca_topic_score_codex":0.003942627,"about_ca_topic_score_gemma":0.0053638862,"teacher_disagreement_score":0.997464,"about_ca_system_score_codex":0.0010031187,"about_ca_system_score_gemma":0.0018961881,"threshold_uncertainty_score":0.019650698},"labels":[],"label_agreement":null},{"id":"W4408463271","doi":"10.1007/s10664-025-10632-2","title":"Empathy, self-determination and motivation: moderating diversity for enhanced performance in software development teams","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Gender Diversity and Inequality","field":"Social Sciences","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Empathy; Diversity (politics); Psychology; Applied psychology; Knowledge management; Computer science; Social psychology; Sociology","score_opus":0.04744867423182276,"score_gpt":0.284422745266576,"score_spread":0.23697407103475326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408463271","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99904937,0.00003361214,0.00010741481,0.00010531392,0.0000034676902,0.000004721733,0.0000061071437,0.0000011188755,0.0006887955],"genre_scores_gemma":[0.9996847,0.000009306499,0.00008829515,0.000020547757,0.000003114529,0.000007816212,0.0000046699756,0.00000166462,0.00017992621],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9939535,0.0038070616,0.00020584058,0.0005416948,0.0006053575,0.0008865868],"domain_scores_gemma":[0.9667528,0.019478489,0.0057996893,0.001274922,0.0010865248,0.0056076082],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006077168,0.0003163974,0.00040550437,0.0006956195,0.002249939,0.002773916,0.00061634183,0.0012384787,0.004333267],"category_scores_gemma":[0.023350475,0.00026549873,0.00044825065,0.0004119111,0.0018184736,0.0010425012,0.003619482,0.0018629726,0.00027300077],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011513208,0.0030825355,0.9402056,0.000120347024,0.00022307964,0.00023666039,0.022929966,0.00056999584,0.003441179,0.0020094581,0.00034189757,0.025688067],"study_design_scores_gemma":[0.000046517238,0.00049296033,0.98645264,0.00006338682,0.00007725693,0.000097000186,0.009263568,0.0008219226,0.00074453565,0.0014437935,0.00047429572,0.000022075097],"about_ca_topic_score_codex":0.0015347148,"about_ca_topic_score_gemma":0.0025732766,"teacher_disagreement_score":0.006077168,"about_ca_system_score_codex":0.0009776778,"about_ca_system_score_gemma":0.0020919612,"threshold_uncertainty_score":0.03213954},"labels":[],"label_agreement":null},{"id":"W4409168587","doi":"10.1007/s10664-024-10607-9","title":"How far are app secrets from being stolen? a case study on android","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada; City University of Hong Kong","keywords":"Android (operating system); Computer science; Mobile apps; Computer security; Operating system; Internet privacy; World Wide Web","score_opus":0.014935987453394393,"score_gpt":0.27402721746446645,"score_spread":0.25909123001107204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409168587","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9932318,0.00016469309,0.0013782582,0.00067345443,0.000010737125,0.000040247156,0.00010545383,0.00001665263,0.004378778],"genre_scores_gemma":[0.9952024,0.00028124894,0.001811128,0.00008848338,0.000008105275,0.000017121873,0.00008288323,0.000017614742,0.0024910376],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9983758,0.00073929626,0.00008286823,0.00014861544,0.00042108208,0.00023240218],"domain_scores_gemma":[0.9766663,0.019354096,0.0014844431,0.00074964395,0.001264888,0.00048073535],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016896888,0.00047700419,0.00023397425,0.0013579463,0.003142833,0.0012927362,0.0011128167,0.0024763825,0.002259515],"category_scores_gemma":[0.013861073,0.00031576154,0.00032177535,0.0010708557,0.0017990343,0.0029080415,0.0012760152,0.0015003642,0.0004533043],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011759705,0.005072011,0.42301127,0.0012664855,0.00017068825,0.10078023,0.21191977,0.009301615,0.012037922,0.02446183,0.013185811,0.19761637],"study_design_scores_gemma":[0.00013462384,0.0022604382,0.2758181,0.0013983222,0.00030088273,0.07782618,0.50203615,0.041437,0.023754166,0.012621204,0.062157627,0.00025542232],"about_ca_topic_score_codex":0.011203611,"about_ca_topic_score_gemma":0.025237994,"teacher_disagreement_score":0.011203611,"about_ca_system_score_codex":0.001189944,"about_ca_system_score_gemma":0.0009927001,"threshold_uncertainty_score":0.022276819},"labels":[],"label_agreement":null},{"id":"W4409405452","doi":"10.1007/s10664-025-10656-8","title":"Logging requirement for continuous auditing of responsible machine learning-based applications","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Audit; Logging; Computer science; Business; Accounting; Forestry; Geography","score_opus":0.015888302818632512,"score_gpt":0.29317120968724425,"score_spread":0.27728290686861173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409405452","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18990885,0.00039672,0.76855415,0.003924226,0.0006581001,0.00074272364,0.00068337866,0.025185812,0.009946037],"genre_scores_gemma":[0.95828974,0.00006778966,0.037793417,0.000508827,0.00012053632,0.00019264682,0.00028618955,0.0005954651,0.002145411],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9766692,0.0048592994,0.0029653155,0.0029279778,0.010097073,0.0024810531],"domain_scores_gemma":[0.7311523,0.094423585,0.019051258,0.12346285,0.026731413,0.005178599],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014858161,0.0008848664,0.0015019673,0.0015283201,0.0016684089,0.004363629,0.0035425373,0.0032333098,0.007536263],"category_scores_gemma":[0.15776864,0.0010795406,0.0009190487,0.0007676292,0.0028100917,0.0077693127,0.005125735,0.005919969,0.0029683458],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005407839,0.0020758547,0.061693266,0.0014585252,0.000305903,0.0045361165,0.0018838714,0.11213192,0.14049277,0.23316224,0.028572597,0.40827915],"study_design_scores_gemma":[0.0002157181,0.0007173778,0.012571949,0.00037020197,0.00014540838,0.0026281325,0.00047983002,0.7283096,0.09796417,0.14224945,0.014174184,0.00017389284],"about_ca_topic_score_codex":0.0011020298,"about_ca_topic_score_gemma":0.00079902995,"teacher_disagreement_score":0.014858161,"about_ca_system_score_codex":0.0012779988,"about_ca_system_score_gemma":0.0060882494,"threshold_uncertainty_score":0.07857835},"labels":[],"label_agreement":null},{"id":"W4409405484","doi":"10.1007/s10664-025-10651-z","title":"Predicting the understandability of computational notebooks through code metrics analysis","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute","funders":"","keywords":"Computer science; Programming language; Code (set theory)","score_opus":0.03729922534070134,"score_gpt":0.3189262083606235,"score_spread":0.2816269830199222,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409405484","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98981047,0.00011607091,0.0082276305,0.000078207595,0.000006786249,0.000028223394,0.0005652657,0.0003418314,0.00082551595],"genre_scores_gemma":[0.98748463,0.000083907325,0.010010483,0.000013829696,0.0000061965475,0.000026549107,0.0014907521,0.00009305151,0.00079062895],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9987047,0.0003083187,0.00010219162,0.0002254568,0.000583606,0.00007572612],"domain_scores_gemma":[0.95273274,0.030813415,0.0069952193,0.0026177894,0.006072269,0.0007686264],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0013753106,0.0005328371,0.00023920949,0.0036603597,0.0002773339,0.0013034137,0.00045914063,0.0006452331,0.0011697988],"category_scores_gemma":[0.04871589,0.00020501569,0.00036082647,0.0023303186,0.00034820303,0.0022713826,0.0006280015,0.00068229187,0.00050967094],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005330917,0.0004741016,0.6691598,0.00029050373,0.00019758783,0.0003175203,0.001443113,0.0476392,0.01891554,0.0022339418,0.0029730476,0.25582254],"study_design_scores_gemma":[0.00003951288,0.000847526,0.507492,0.0000915329,0.00010853278,0.00030217308,0.0009135983,0.45934057,0.022479469,0.0048443666,0.0034624878,0.00007831971],"about_ca_topic_score_codex":0.008177218,"about_ca_topic_score_gemma":0.011721239,"teacher_disagreement_score":0.9986247,"about_ca_system_score_codex":0.0007374247,"about_ca_system_score_gemma":0.0006207707,"threshold_uncertainty_score":0.016259253},"labels":[],"label_agreement":null},{"id":"W4409405816","doi":"10.1007/s10664-025-10648-8","title":"Opportunities and security risks of technical leverage: A replication study on the NPM ecosystem","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Information and Cyber Security","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; Concordia University","funders":"","keywords":"Leverage (statistics); Replication (statistics); Business; Computer science; Computer security; Medicine","score_opus":0.07617329618644243,"score_gpt":0.3188694552498021,"score_spread":0.24269615906335967,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409405816","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.994079,0.00006641548,0.0010483265,0.00018199811,0.000008802444,0.00015296573,0.00014311595,0.000016805425,0.0043026707],"genre_scores_gemma":[0.9975426,0.00003356468,0.0010684141,0.00008275945,0.000009150991,0.00021495584,0.00012810995,0.000011155883,0.00090927933],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9949705,0.0024310818,0.00026976698,0.0007550247,0.0011726093,0.0004011352],"domain_scores_gemma":[0.90735316,0.045628946,0.0114804795,0.025949238,0.0073769884,0.0022112187],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01335824,0.00049668935,0.0006299228,0.0016725853,0.0025706945,0.002863468,0.0020762102,0.0019399694,0.006853362],"category_scores_gemma":[0.07933564,0.00047980805,0.0008762338,0.0013970567,0.002324623,0.0060817576,0.0035804252,0.002837022,0.001025781],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030422893,0.009882088,0.86845773,0.00038051073,0.00063829566,0.0021505172,0.024288597,0.0017423801,0.0033884696,0.015921481,0.0027631063,0.0673446],"study_design_scores_gemma":[0.00082534197,0.0047076466,0.895882,0.00035905658,0.0006785595,0.0014753018,0.04556083,0.013229396,0.0036001457,0.02093411,0.012600893,0.00014667423],"about_ca_topic_score_codex":0.0056591835,"about_ca_topic_score_gemma":0.0049981065,"teacher_disagreement_score":0.98664176,"about_ca_system_score_codex":0.0014699777,"about_ca_system_score_gemma":0.0028508964,"threshold_uncertainty_score":0.07064593},"labels":[],"label_agreement":null},{"id":"W4409441986","doi":"10.1007/s10664-025-10655-9","title":"Predicting long time contributors with knowledge units of programming languages: an empirical study","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Empirical research; Programming language; Natural language processing; Data science; Artificial intelligence; Statistics; Mathematics","score_opus":0.017723677763842167,"score_gpt":0.3199314896089665,"score_spread":0.30220781184512435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409441986","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"incentives","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"incentives","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99918467,0.00004571381,0.0003041425,0.00006830019,0.0000046477226,0.000009010585,0.00008072743,0.000007384114,0.00029541907],"genre_scores_gemma":[0.9975005,0.00007342078,0.0004092199,0.000026880867,0.000015088724,0.000018636656,0.0002543497,0.000014731578,0.0016872106],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99784946,0.00094497314,0.00014663057,0.0003735636,0.00046374946,0.00022157791],"domain_scores_gemma":[0.8323547,0.11714985,0.022889588,0.0074130306,0.0069166566,0.013276225],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.003626236,0.0003620716,0.0003419942,0.0021263834,0.0010867858,0.0017946647,0.0013780391,0.0015549969,0.0067471876],"category_scores_gemma":[0.057838365,0.0005234166,0.00040958446,0.0014814213,0.0006475766,0.003115134,0.0016617001,0.0022511124,0.0016359518],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023950906,0.0011620885,0.9873892,0.000017224385,0.000042657153,0.00015280602,0.0020720745,0.00029263351,0.00027127433,0.0002207273,0.00035477863,0.0077849953],"study_design_scores_gemma":[0.00004918973,0.00058645144,0.9766533,0.00004047132,0.000109683824,0.00044051927,0.0063989027,0.0122944545,0.0008750968,0.00077310845,0.0017409769,0.000037956175],"about_ca_topic_score_codex":0.008350407,"about_ca_topic_score_gemma":0.010488621,"teacher_disagreement_score":0.9978736,"about_ca_system_score_codex":0.00054392515,"about_ca_system_score_gemma":0.001004037,"threshold_uncertainty_score":0.022571564},"labels":[],"label_agreement":null},{"id":"W4409608564","doi":"10.1007/s10664-025-10633-1","title":"Impact of extensions on browser performance: An empirical study on google chrome","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Queen's University; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Empirical research; World Wide Web; Computer science; Business; Mathematics","score_opus":0.03426352058869829,"score_gpt":0.36338248516094646,"score_spread":0.32911896457224815,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409608564","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9992987,0.000035169254,0.000026844786,0.000013348945,0.000001098722,0.00000360927,0.00005668859,0.000017838418,0.00054681377],"genre_scores_gemma":[0.9991542,0.000036929483,0.00013230633,0.000012455269,0.000004441309,0.0000033687695,0.00022249398,0.000015526546,0.0004182949],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99849534,0.00041932787,0.00008441467,0.00020810703,0.0005435319,0.00024934552],"domain_scores_gemma":[0.95974964,0.026475245,0.0041555665,0.0023864415,0.0046355603,0.0025975737],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017919289,0.00040309326,0.00029155792,0.0012650386,0.000590431,0.0017218293,0.0005120182,0.0007405186,0.0019710672],"category_scores_gemma":[0.02233327,0.00020182997,0.00033267087,0.0018180453,0.0006697505,0.0025566772,0.0005102329,0.0013983413,0.00057736103],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017317514,0.0046101515,0.9472607,0.00017554412,0.00018900499,0.00058383215,0.0024797844,0.0035468396,0.0061716354,0.0006561125,0.002189572,0.030405173],"study_design_scores_gemma":[0.000048995433,0.0011674119,0.9849744,0.000033252913,0.00013377774,0.00027445614,0.0021659639,0.007830023,0.0019470769,0.00018691472,0.0011960266,0.000041679075],"about_ca_topic_score_codex":0.01303917,"about_ca_topic_score_gemma":0.013376428,"teacher_disagreement_score":0.01303917,"about_ca_system_score_codex":0.00077142444,"about_ca_system_score_gemma":0.00082823506,"threshold_uncertainty_score":0.02592653},"labels":[],"label_agreement":null},{"id":"W4409833316","doi":"10.1007/s10664-025-10654-w","title":"Towards understanding code review practices for infrastructure-as-code: An empirical study on OpenStack projects","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Code (set theory); Empirical research; Operating system; Software engineering; Programming language","score_opus":0.09559856838119425,"score_gpt":0.4004029999918837,"score_spread":0.3048044316106895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409833316","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99733174,0.0000827148,0.0011253624,0.00025949412,0.0000029728656,0.000041554034,0.000037090416,0.000026181995,0.0010930181],"genre_scores_gemma":[0.9980306,0.000081705446,0.0012727655,0.00005358732,0.000003511593,0.000036581696,0.000074820484,0.000030716517,0.00041560357],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9801763,0.008393259,0.0015752295,0.0018813334,0.0064701196,0.0015036779],"domain_scores_gemma":[0.59181386,0.2636143,0.081480026,0.012187436,0.04276207,0.008142234],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.020582324,0.0002849917,0.00030924953,0.005222726,0.0021831112,0.0041922648,0.0015642219,0.0015222467,0.0015646338],"category_scores_gemma":[0.21847637,0.0005376847,0.0002995178,0.0041371947,0.0031899256,0.006960868,0.0031810892,0.0021422817,0.00036541946],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017839101,0.0010415369,0.7845932,0.0003531408,0.00007499945,0.0005322256,0.14598554,0.0010738045,0.0040869876,0.0025795158,0.0011783362,0.05832233],"study_design_scores_gemma":[0.000025629264,0.00055620505,0.84495616,0.00037979838,0.00006111745,0.00063408574,0.13662171,0.005839375,0.002241596,0.0018287525,0.0067664622,0.00008918609],"about_ca_topic_score_codex":0.014134202,"about_ca_topic_score_gemma":0.025198154,"teacher_disagreement_score":0.9794177,"about_ca_system_score_codex":0.004184602,"about_ca_system_score_gemma":0.008717596,"threshold_uncertainty_score":0.108851016},"labels":[],"label_agreement":null},{"id":"W4410602514","doi":"10.1007/s10664-025-10665-7","title":"Correction to: Utilization of pre-trained language models for adapter-based knowledge transfer in software engineering","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Okanagan University College; University of British Columbia, Okanagan Campus; University of British Columbia","funders":"","keywords":"Adapter (computing); Computer science; Software engineering; Natural language processing; Artificial intelligence; Operating system","score_opus":0.02983421994874074,"score_gpt":0.293690710703116,"score_spread":0.26385649075437523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410602514","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0025932372,0.0005540893,0.0073022214,0.04275146,0.93332475,0.00010447198,0.0066899573,0.004041347,0.002638486],"genre_scores_gemma":[0.2484927,0.004996076,0.06995574,0.06627736,0.23312363,0.0015199756,0.024647566,0.012827752,0.33815917],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99201524,0.000912549,0.0022166916,0.0016283396,0.0024828857,0.000744204],"domain_scores_gemma":[0.83854115,0.041768536,0.006720295,0.017843913,0.09109244,0.004033608],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044439915,0.002221977,0.0032845235,0.0044436646,0.0037106676,0.0041010003,0.0047395104,0.006869931,0.13621053],"category_scores_gemma":[0.16596168,0.0011678532,0.0016677767,0.004995053,0.002255776,0.0039021035,0.0040530884,0.00869756,0.068383776],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025017638,0.000049468385,0.00057598675,0.0004065355,0.00007019706,0.0008493999,0.00028366517,0.00028425854,0.000591884,0.0021742429,0.96268547,0.031778682],"study_design_scores_gemma":[0.00027189223,0.000119829725,0.010332727,0.0006460767,0.00012637473,0.0032144063,0.000939454,0.005710281,0.0067967484,0.010606909,0.9609492,0.00028606295],"about_ca_topic_score_codex":0.009948956,"about_ca_topic_score_gemma":0.012817865,"teacher_disagreement_score":0.13621053,"about_ca_system_score_codex":0.0036152985,"about_ca_system_score_gemma":0.004129287,"threshold_uncertainty_score":0.45566964},"labels":[],"label_agreement":null},{"id":"W4411278626","doi":"10.1007/s10664-025-10682-6","title":"On the need to monitor continuous integration practices","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science","score_opus":0.0242274045556198,"score_gpt":0.31203987733416977,"score_spread":0.28781247277855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411278626","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60209394,0.011252043,0.09521682,0.19475959,0.001034217,0.0002741747,0.00040757546,0.0003887686,0.09457291],"genre_scores_gemma":[0.95602304,0.0031753758,0.030892877,0.0065981243,0.00035510227,0.00015188903,0.0001576346,0.000079716905,0.002566151],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9685135,0.014468185,0.0020455138,0.0018215086,0.012186001,0.00096533285],"domain_scores_gemma":[0.45842534,0.426747,0.035504494,0.0248974,0.049325846,0.0050999406],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.037488174,0.00038594048,0.00044305623,0.0028429555,0.0015289707,0.0054521337,0.0023448095,0.0038863758,0.0047080996],"category_scores_gemma":[0.24359554,0.00043798672,0.00025829586,0.0028393823,0.0052228873,0.015543866,0.003406978,0.0047054417,0.0007422539],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049278885,0.00080386555,0.21072371,0.0013982478,0.00014767714,0.00054250524,0.018838836,0.0039681625,0.004494243,0.1180923,0.02101305,0.61948466],"study_design_scores_gemma":[0.00013483461,0.0014611034,0.47188717,0.005751243,0.00022995648,0.0020309319,0.075138785,0.022416074,0.0051915455,0.3024354,0.1130174,0.0003056595],"about_ca_topic_score_codex":0.005670696,"about_ca_topic_score_gemma":0.012431862,"teacher_disagreement_score":0.96251184,"about_ca_system_score_codex":0.0032197698,"about_ca_system_score_gemma":0.006212004,"threshold_uncertainty_score":0.19825882},"labels":[],"label_agreement":null},{"id":"W4411440374","doi":"10.1007/s10664-025-10661-x","title":"Supporting multi-dimensional unit test classification","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Unit testing; Test (biology); Unit (ring theory); Artificial intelligence; Pattern recognition (psychology); Geology; Psychology; Operating system; Mathematics education","score_opus":0.04423106979949689,"score_gpt":0.34067254859068613,"score_spread":0.29644147879118926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411440374","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27429515,0.001407177,0.68895686,0.0008348183,0.00017895999,0.00057920144,0.008179392,0.022470292,0.0030980934],"genre_scores_gemma":[0.5206039,0.0002100652,0.45826927,0.00029761464,0.00008332834,0.00033518326,0.018589443,0.0005689745,0.0010422777],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9909928,0.0014498674,0.0015204047,0.0022501606,0.003304785,0.00048198632],"domain_scores_gemma":[0.95365673,0.022926288,0.005153484,0.008529459,0.008679735,0.001054351],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059470935,0.0014203084,0.0020008322,0.006860353,0.0009260309,0.0031925468,0.0034195762,0.0020325785,0.0012118662],"category_scores_gemma":[0.040013388,0.00045919864,0.0014972413,0.005032581,0.000911596,0.0035971196,0.0026557534,0.0021468091,0.0011141459],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007978002,0.00096463197,0.14386603,0.0007458571,0.000261658,0.00076711003,0.00088915986,0.03793439,0.019581351,0.0043029436,0.014727284,0.77516174],"study_design_scores_gemma":[0.00006356014,0.00018190473,0.018067634,0.000100968784,0.00008555252,0.00059177063,0.00038503736,0.9374019,0.023374945,0.013646349,0.0060276985,0.00007269984],"about_ca_topic_score_codex":0.0058789174,"about_ca_topic_score_gemma":0.01080994,"teacher_disagreement_score":0.006860353,"about_ca_system_score_codex":0.0012497718,"about_ca_system_score_gemma":0.0019212636,"threshold_uncertainty_score":0.031451643},"labels":[],"label_agreement":null},{"id":"W4411538180","doi":"10.1007/s10664-025-10669-3","title":"A comprehensive study of machine learning techniques for log-based anomaly detection","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Mitacs; Canada Research Chairs","keywords":"Anomaly detection; Anomaly (physics); Computer science; Machine learning; Artificial intelligence; Data mining; Physics","score_opus":0.01575046496258971,"score_gpt":0.27854004777971597,"score_spread":0.2627895828171263,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411538180","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044518735,0.018550511,0.9280307,0.0019042061,0.0001929804,0.00017496437,0.0003706176,0.0030836284,0.0031736735],"genre_scores_gemma":[0.54107016,0.011690054,0.44256383,0.00051311555,0.0004650464,0.0003386842,0.0010636474,0.000303827,0.0019915737],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99279445,0.0025459088,0.0005003057,0.0012808393,0.0026598847,0.00021855361],"domain_scores_gemma":[0.9744848,0.01740716,0.002493119,0.0026944939,0.0026915763,0.00022882805],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0078033446,0.0023576692,0.0014267009,0.003071209,0.0005238923,0.0017439589,0.0022776544,0.0017010926,0.0009222004],"category_scores_gemma":[0.030377733,0.0007583144,0.001408352,0.002526816,0.0009385429,0.0038040064,0.0012655442,0.0039240494,0.0009302447],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017017954,0.00032126805,0.013000008,0.0010689829,0.00042632068,0.00009975416,0.0001429396,0.216382,0.00840158,0.007334358,0.004291807,0.74836075],"study_design_scores_gemma":[0.000014955614,0.00030534557,0.006507696,0.00019365828,0.00006889378,0.00035359414,0.000056823363,0.96610695,0.01070637,0.00922205,0.0064115133,0.000052155934],"about_ca_topic_score_codex":0.001671814,"about_ca_topic_score_gemma":0.0016929012,"teacher_disagreement_score":0.0078033446,"about_ca_system_score_codex":0.0013794656,"about_ca_system_score_gemma":0.0012988143,"threshold_uncertainty_score":0.041268528},"labels":[],"label_agreement":null},{"id":"W4412152791","doi":"10.1007/s10664-025-10629-x","title":"Guiding principles for mixed methods research in software engineering","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada; Monash University; University of Victoria","keywords":"Computer science; Software engineering; Systems engineering; Engineering","score_opus":0.15558234056282055,"score_gpt":0.44330466850829364,"score_spread":0.2877223279454731,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412152791","genre_codex":"methods","genre_gemma":"methods","domain_codex":"methods","domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":"methods","prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011241953,0.0021647464,0.97449017,0.010428149,0.00035285729,0.0023781734,0.00012341238,0.00014715805,0.008791169],"genre_scores_gemma":[0.022877967,0.0010227553,0.9646105,0.0021441714,0.00018806155,0.008251462,0.000044968456,0.0000752258,0.0007848508],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.4538896,0.4843882,0.017619103,0.009984115,0.032288413,0.0018304569],"domain_scores_gemma":[0.47039464,0.43975976,0.014771192,0.034515627,0.03704202,0.0035167865],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.465583,0.0029648847,0.004069092,0.007016927,0.0069595138,0.016168384,0.007387648,0.009510363,0.0043757465],"category_scores_gemma":[0.31547827,0.003668671,0.0043962086,0.0053606867,0.040863484,0.010227428,0.0100828735,0.017801056,0.0033719316],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005685562,0.00007433227,0.00035836015,0.0012730671,0.00009279154,0.000057624373,0.0027861886,0.0018266479,0.00023609887,0.9692077,0.0022575448,0.021772863],"study_design_scores_gemma":[0.00021506508,0.00016729491,0.00025771116,0.0023958166,0.00007025678,0.00013346998,0.0006046523,0.006129259,0.0010691962,0.9409973,0.047881465,0.00007852179],"about_ca_topic_score_codex":0.0036246774,"about_ca_topic_score_gemma":0.0034094998,"teacher_disagreement_score":0.53441703,"about_ca_system_score_codex":0.011361227,"about_ca_system_score_gemma":0.01915759,"threshold_uncertainty_score":0.6590313},"labels":[],"label_agreement":null},{"id":"W4412825323","doi":"10.1007/s10664-025-10705-2","title":"An Empirical Investigation on the Challenges in Scientific Workflow Systems Development","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Workflow; Computer science; Systems engineering; Data science; Empirical research; Software engineering; Engineering; Database; Mathematics","score_opus":0.24292159394730398,"score_gpt":0.3787638175672227,"score_spread":0.13584222361991874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412825323","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98115516,0.00028550503,0.0034089317,0.0030299982,0.00002837841,0.00015958039,0.00009892294,0.000013897342,0.0118195135],"genre_scores_gemma":[0.99767715,0.00010326355,0.0014233162,0.00015324178,0.000011656235,0.00007128185,0.00006397493,0.0000073667247,0.0004888496],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.97604126,0.01446864,0.0017370193,0.0012872076,0.0050314986,0.0014344447],"domain_scores_gemma":[0.46371326,0.4698483,0.028073907,0.009203776,0.023637408,0.005523345],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0361639,0.00031754057,0.00029108074,0.0021436592,0.002962874,0.005680424,0.0013669423,0.0016954804,0.004646744],"category_scores_gemma":[0.24701777,0.00041769727,0.00039434698,0.0036067376,0.0035143727,0.010618493,0.0027764565,0.0030360592,0.00052964734],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008665827,0.0037424488,0.76389587,0.0008825165,0.00011692047,0.0013228706,0.06639844,0.0042681983,0.0014166407,0.059900984,0.0044046342,0.0927838],"study_design_scores_gemma":[0.00027020555,0.0023230764,0.55070794,0.001445091,0.0001857365,0.001309124,0.31174392,0.03712622,0.0037155696,0.052357588,0.038665734,0.0001498949],"about_ca_topic_score_codex":0.0050341,"about_ca_topic_score_gemma":0.006596909,"teacher_disagreement_score":0.9638361,"about_ca_system_score_codex":0.0034131748,"about_ca_system_score_gemma":0.009685759,"threshold_uncertainty_score":0.19125527},"labels":[],"label_agreement":null},{"id":"W4413087881","doi":"10.1007/s10664-025-10685-3","title":"On state reverting in solidity smart contracts: Developer practices, fault categorization, and tool evaluation","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Solidity; Categorization; Computer science; State (computer science); Fault (geology); Mean reversion; Engineering; Artificial intelligence; Economics; Econometrics; Programming language","score_opus":0.019389849230910663,"score_gpt":0.29830760325301664,"score_spread":0.278917754022106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413087881","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95913386,0.00035525855,0.034003906,0.00095536927,0.000024383486,0.00015505939,0.00013188692,0.00023063563,0.005009679],"genre_scores_gemma":[0.98604894,0.000078227255,0.012829007,0.000065288616,0.000005964884,0.000051522493,0.000120936325,0.000055169658,0.000744854],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.94286895,0.036905803,0.003149201,0.0029472099,0.012263095,0.0018658221],"domain_scores_gemma":[0.4280976,0.4703194,0.025958221,0.037473362,0.035303745,0.0028477476],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07718959,0.00062661566,0.000605365,0.004752953,0.0020384504,0.0039575966,0.0021476184,0.001735403,0.0028013266],"category_scores_gemma":[0.31493145,0.00047776,0.00041048817,0.0034484926,0.0033587948,0.011167526,0.0038694884,0.0022816586,0.00044041334],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020210778,0.0031895223,0.39228383,0.0005863275,0.00027178088,0.00030933623,0.028793849,0.033640712,0.004210719,0.0409616,0.0042022127,0.4895291],"study_design_scores_gemma":[0.0005005216,0.004648585,0.24258605,0.0011850414,0.0004372773,0.00078216277,0.036527127,0.55319154,0.03260843,0.11235509,0.01471818,0.00045995365],"about_ca_topic_score_codex":0.007511491,"about_ca_topic_score_gemma":0.014220286,"teacher_disagreement_score":0.92281044,"about_ca_system_score_codex":0.0048546614,"about_ca_system_score_gemma":0.004674347,"threshold_uncertainty_score":0.40822244},"labels":[],"label_agreement":null},{"id":"W4413219439","doi":"10.1007/s10664-025-10693-3","title":"Adversarial attack classification and robustness testing for large language models for code","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Robustness (evolution); Adversarial system; Artificial intelligence; Programming language; Biology","score_opus":0.04674644636670435,"score_gpt":0.3271197308610341,"score_spread":0.28037328449432974,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413219439","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06454762,0.0004132369,0.93037933,0.0010252743,0.00007119097,0.00008522037,0.00023181607,0.0017185814,0.0015276694],"genre_scores_gemma":[0.88829494,0.00022099905,0.10639226,0.0003618463,0.00021599329,0.00017171992,0.0010794984,0.0004454501,0.002817257],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9926466,0.0042414917,0.00030625417,0.00108961,0.0012752245,0.00044076698],"domain_scores_gemma":[0.9211485,0.067526765,0.0025359413,0.006486217,0.0015719859,0.000730628],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010190294,0.001596007,0.0013724373,0.0023218421,0.00083303504,0.0018252487,0.0021473793,0.002283921,0.0030585798],"category_scores_gemma":[0.066360235,0.0007081736,0.001791338,0.00091205485,0.0028915657,0.0034683181,0.004320963,0.0044041863,0.0007008571],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036416345,0.0002279033,0.0047665965,0.0001381107,0.00019016101,0.00015352022,0.00017473519,0.8483362,0.003438636,0.04318233,0.004530644,0.09449698],"study_design_scores_gemma":[0.0000063716643,0.000018571916,0.00021601972,0.0000065055615,0.00000578145,0.000014619364,0.0000074226446,0.9843593,0.00059731165,0.014628591,0.00013384283,0.0000056210492],"about_ca_topic_score_codex":0.0027018995,"about_ca_topic_score_gemma":0.002597441,"teacher_disagreement_score":0.010190294,"about_ca_system_score_codex":0.0017258336,"about_ca_system_score_gemma":0.0014210044,"threshold_uncertainty_score":0.053892076},"labels":[],"label_agreement":null},{"id":"W4413308446","doi":"10.1007/s10664-025-10707-0","title":"Towards understanding the challenges of bug localization in deep learning systems","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Deep learning; Data science; Artificial intelligence; Software engineering","score_opus":0.042013341184440375,"score_gpt":0.2910416336842223,"score_spread":0.24902829249978192,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413308446","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11640395,0.0024975212,0.86499465,0.011404836,0.00012660715,0.00007291309,0.00033422455,0.000788277,0.003377018],"genre_scores_gemma":[0.87095386,0.0011758065,0.124076515,0.0007706493,0.0001926187,0.0000641817,0.0003710143,0.00018566758,0.0022095712],"study_design_codex":"simulation_or_modeling","study_design_gemma":"qualitative","domain_scores_codex":[0.9964162,0.0015419251,0.00022376298,0.0006759693,0.0007522824,0.00038987925],"domain_scores_gemma":[0.9533638,0.03423564,0.0037107451,0.003877431,0.0039573465,0.0008549324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067130523,0.00090145227,0.0011963465,0.0017783119,0.0008003928,0.0050304276,0.0021304039,0.0029662368,0.0024875957],"category_scores_gemma":[0.06177506,0.0009890986,0.0006377535,0.0012431311,0.0026997481,0.01264092,0.0041648815,0.00580579,0.0004492895],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003445415,0.00045262117,0.04120562,0.000695244,0.00023250844,0.00028813956,0.0016888736,0.41670752,0.004926101,0.23345998,0.0085484125,0.29145044],"study_design_scores_gemma":[0.000008369932,0.000025649511,0.0011199042,0.000058703525,0.00001414218,0.000035258327,0.00020193304,0.77353334,0.0008193106,0.22290859,0.0012627535,0.000011988563],"about_ca_topic_score_codex":0.008028561,"about_ca_topic_score_gemma":0.008116927,"teacher_disagreement_score":0.008028561,"about_ca_system_score_codex":0.0018559756,"about_ca_system_score_gemma":0.002961522,"threshold_uncertainty_score":0.035502434},"labels":[],"label_agreement":null},{"id":"W4414474108","doi":"10.1007/s10664-025-10667-5","title":"Correction to: Assessing the adoption of security policies by developers in terraform across different cloud providers","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Information and Cyber Security","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Cloud computing; Cloud service provider; Security policy; Cloud computing security","score_opus":0.010958199298789092,"score_gpt":0.2911385402859514,"score_spread":0.28018034098716227,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414474108","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00050624576,0.00033878218,0.001036837,0.09246165,0.89377725,0.00006222587,0.007643543,0.0013414349,0.0028321098],"genre_scores_gemma":[0.049071096,0.00493445,0.01399302,0.19160874,0.26155278,0.0011591839,0.023045164,0.008378821,0.44625667],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98508865,0.0017636853,0.0030009584,0.002290088,0.0063554375,0.0015011791],"domain_scores_gemma":[0.73615915,0.04875078,0.011008203,0.01795322,0.17792808,0.008200548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008793398,0.0025214914,0.0032367092,0.00950067,0.0062106526,0.0075185797,0.005832897,0.012002802,0.124651216],"category_scores_gemma":[0.20948552,0.0020670532,0.002002263,0.009405989,0.004066616,0.004453607,0.005007197,0.014138237,0.066404626],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000364693,0.00000856518,0.00019740028,0.000117028314,0.000015610354,0.000102653685,0.00008190653,0.00006432578,0.000044958655,0.000541239,0.99573034,0.0030594065],"study_design_scores_gemma":[0.00016356919,0.000046401386,0.005799183,0.0008280426,0.00008076501,0.00040656596,0.00064144476,0.0008218146,0.0006862711,0.002187509,0.9881543,0.00018417537],"about_ca_topic_score_codex":0.08557467,"about_ca_topic_score_gemma":0.08386606,"teacher_disagreement_score":0.124651216,"about_ca_system_score_codex":0.008696767,"about_ca_system_score_gemma":0.016111702,"threshold_uncertainty_score":0.41699988},"labels":[],"label_agreement":null},{"id":"W4414589104","doi":"10.1007/s10664-025-10717-y","title":"Towards understanding the impact of data bugs on deep learning models in software engineering","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Dalhousie University","funders":"","keywords":"Deep learning; Overfitting; Generalizability theory; Data pre-processing; Software quality; Leverage (statistics); Preprocessor; Metric (unit); Software bug; Software","score_opus":0.09277203969085854,"score_gpt":0.34203059390168095,"score_spread":0.2492585542108224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414589104","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44974697,0.0028287126,0.5333462,0.008423022,0.0002009618,0.00005858683,0.0004070438,0.001048957,0.0039395452],"genre_scores_gemma":[0.9653669,0.0005115622,0.03231659,0.00031200782,0.00007475136,0.000026768892,0.0002349541,0.00014225715,0.0010141599],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.996784,0.0016764044,0.00018395834,0.00044289522,0.0006351525,0.00027763774],"domain_scores_gemma":[0.8846455,0.09872138,0.0051708003,0.0051543806,0.0053420872,0.0009657761],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009754561,0.0009840745,0.0009062155,0.0013044184,0.0005024909,0.0031018576,0.0016104279,0.001975525,0.0024209816],"category_scores_gemma":[0.12418007,0.00093298923,0.0005857886,0.0011112054,0.0016055813,0.010612553,0.0024051664,0.005790864,0.0002627426],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043611124,0.00037596142,0.035379935,0.0002871477,0.00018765785,0.00012997551,0.00038251217,0.7907699,0.0022905818,0.05901787,0.0030780686,0.10766425],"study_design_scores_gemma":[0.000005153065,0.000018827875,0.0006075088,0.000019652343,0.000010890816,0.0000067200626,0.000022863087,0.9741416,0.00039384872,0.024647178,0.00012125744,0.0000045034863],"about_ca_topic_score_codex":0.011293578,"about_ca_topic_score_gemma":0.011947529,"teacher_disagreement_score":0.011293578,"about_ca_system_score_codex":0.0024504869,"about_ca_system_score_gemma":0.0021078335,"threshold_uncertainty_score":0.05158764},"labels":[],"label_agreement":null},{"id":"W4414622738","doi":"10.1007/s10664-025-10731-0","title":"DeepCodeProbe: Evaluating Code Representation Quality in Models Trained on Code","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); Polytechnique Montréal","funders":"Fonds de Recherche du Québec - Santé; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Representation (politics); Software quality; Code (set theory); Quality (philosophy); Code smell; Source code; Software; Static program analysis","score_opus":0.11928940518778033,"score_gpt":0.4194135589370064,"score_spread":0.3001241537492261,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414622738","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.72486836,0.0034139727,0.22329989,0.0015063958,0.0007809361,0.00029167676,0.0059097735,0.03560763,0.0043213083],"genre_scores_gemma":[0.8725873,0.0005239064,0.09947149,0.000653642,0.00008268717,0.00019507603,0.020913005,0.001876179,0.0036966733],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99771225,0.00077950605,0.00015064441,0.0006595314,0.00049807003,0.00019987428],"domain_scores_gemma":[0.98509395,0.009859195,0.00058900216,0.002113223,0.0019538936,0.00039075656],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.004536868,0.0021885647,0.00084718026,0.0016844254,0.00054698193,0.0014158101,0.0025532236,0.0027907172,0.0031020893],"category_scores_gemma":[0.021560043,0.00071277644,0.0012222892,0.0010509909,0.0011018772,0.0035465437,0.0018913839,0.003021563,0.0014038968],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022729915,0.001148936,0.031169748,0.0007853502,0.0008048857,0.0002521475,0.00028333112,0.58698905,0.015808167,0.0028098885,0.035948817,0.32172662],"study_design_scores_gemma":[0.00010078915,0.00027042243,0.0015020577,0.000043560558,0.000058417572,0.000049204165,0.00005531287,0.98896307,0.0061394684,0.0018117634,0.0009886229,0.000017368122],"about_ca_topic_score_codex":0.013716951,"about_ca_topic_score_gemma":0.018611029,"teacher_disagreement_score":0.99546313,"about_ca_system_score_codex":0.0017721774,"about_ca_system_score_gemma":0.0024246639,"threshold_uncertainty_score":0.027274191},"labels":[],"label_agreement":null},{"id":"W4415991171","doi":"10.1007/s10664-025-10732-z","title":"MonoEmbed: Enhancing LLM representations for monolith to microservices decomposition through contrastive learning","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Microservices; Flexibility (engineering); Representation (politics); Decomposition; Adaptation (eye); Rank (graph theory); Benchmark (surveying)","score_opus":0.011455719920588168,"score_gpt":0.32757723681534184,"score_spread":0.31612151689475365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415991171","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027667493,0.00009367337,0.95976585,0.00012601259,0.00006875876,0.00004809267,0.00019249255,0.008421793,0.0036159365],"genre_scores_gemma":[0.37413526,0.000099296136,0.6171227,0.00023513135,0.00004166837,0.00012097825,0.0010548504,0.0011003042,0.0060896985],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997619,0.00006335556,0.000010761035,0.00007554799,0.000060411017,0.000028119277],"domain_scores_gemma":[0.9989453,0.00047288954,0.00006828774,0.000278041,0.0001607336,0.000074813244],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005779309,0.0006756236,0.00033917226,0.0007536059,0.00027663936,0.0010865798,0.0012515628,0.0007792888,0.008936937],"category_scores_gemma":[0.0035283421,0.0002549556,0.0004269911,0.00046622212,0.00035973167,0.0026684846,0.001634338,0.001855851,0.0018464609],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004597937,0.00047238145,0.0012398491,0.00020336742,0.000040596056,0.00013351403,0.00029628357,0.066262595,0.041654952,0.02029104,0.009005501,0.8599402],"study_design_scores_gemma":[0.00003084856,0.00007509657,0.00026875245,0.000021634783,0.000015342002,0.00004275688,0.000052180643,0.9611084,0.018477382,0.0144481845,0.005447347,0.000012035416],"about_ca_topic_score_codex":0.00174018,"about_ca_topic_score_gemma":0.0033714012,"teacher_disagreement_score":0.008936937,"about_ca_system_score_codex":0.0004643356,"about_ca_system_score_gemma":0.00056501763,"threshold_uncertainty_score":0.029897034},"labels":[],"label_agreement":null},{"id":"W4416240214","doi":"10.1007/s10664-025-10700-7","title":"Fuzzing-based mutation testing of C/C++ software in cyber-physical systems","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Kangwon National University; European Space Agency","keywords":"Fuzz testing; Symbolic execution; Code coverage; Process (computing); Software testing; Mutation testing; Mutation; Software; Software bug; Test case","score_opus":0.02328434506771092,"score_gpt":0.2833576778545125,"score_spread":0.26007333278680156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416240214","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8114405,0.00019225274,0.1853522,0.00019819401,0.000023754956,0.000055828335,0.00010820975,0.0012850832,0.0013438881],"genre_scores_gemma":[0.974596,0.000022577233,0.025157347,0.000016967491,0.0000030035155,0.00001155737,0.000040922845,0.000027770018,0.00012383395],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.99802256,0.00075496105,0.00009510081,0.00026955057,0.0007439937,0.00011397034],"domain_scores_gemma":[0.9760105,0.018568156,0.0014128898,0.002074826,0.0017047501,0.0002289185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024873682,0.00048239154,0.0003978202,0.001284132,0.00034127338,0.0006277894,0.0012164233,0.0007849872,0.00087167334],"category_scores_gemma":[0.026751306,0.0002443078,0.00035973525,0.0006426659,0.0008606181,0.0012676222,0.0005497062,0.0007715505,0.00008394055],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009383833,0.00060443045,0.05067787,0.00026718006,0.00017676382,0.00038829062,0.000471059,0.6771694,0.047043473,0.022047488,0.0009372313,0.19927847],"study_design_scores_gemma":[0.000017798353,0.00008425346,0.0046571046,0.000011505515,0.000019797506,0.00008405843,0.000019489848,0.98328847,0.0083233975,0.0034012597,0.00008390314,0.0000089438745],"about_ca_topic_score_codex":0.0049316445,"about_ca_topic_score_gemma":0.005208926,"teacher_disagreement_score":0.0049316445,"about_ca_system_score_codex":0.00082732196,"about_ca_system_score_gemma":0.00095370214,"threshold_uncertainty_score":0.013154566},"labels":[],"label_agreement":null},{"id":"W4416248181","doi":"10.1007/s10664-025-10761-8","title":"SBEST: Spectrum-based fault localization without fault-triggering tests","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Concordia University","funders":"","keywords":"Stack (abstract data type); Software bug; Context (archaeology); TRACE (psycholinguistics); Call stack; Fault (geology); Software regression; Software; Debugging","score_opus":0.01565781770040737,"score_gpt":0.28534965561632986,"score_spread":0.2696918379159225,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416248181","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021138906,0.0001267535,0.92196953,0.000091386675,0.000075850694,0.000093112,0.0004106087,0.05447465,0.0016191663],"genre_scores_gemma":[0.4611116,0.000089193774,0.5311285,0.00017141504,0.00005888853,0.00017767829,0.0011953518,0.003153414,0.0029139263],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997171,0.00066656666,0.00016352188,0.00048386765,0.0012744422,0.00024066046],"domain_scores_gemma":[0.9920861,0.0029980708,0.0007610139,0.0026441007,0.00118563,0.0003250115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016819936,0.001837779,0.001408279,0.0032084328,0.00059793715,0.0011003227,0.003098178,0.0013632729,0.006734475],"category_scores_gemma":[0.009959416,0.0007178409,0.0008673638,0.001345972,0.0011821279,0.0027039896,0.0024659014,0.0014035439,0.0025714003],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030242505,0.000813999,0.009108804,0.00070888846,0.000300243,0.00069709064,0.00032173243,0.10135403,0.06372995,0.021027833,0.021091642,0.77782154],"study_design_scores_gemma":[0.0002564294,0.00048989785,0.0018014507,0.00006207136,0.000102730584,0.0005030567,0.000068212306,0.92137897,0.044135667,0.026951576,0.004184089,0.00006586153],"about_ca_topic_score_codex":0.0021200327,"about_ca_topic_score_gemma":0.002619432,"teacher_disagreement_score":0.006734475,"about_ca_system_score_codex":0.0005111044,"about_ca_system_score_gemma":0.0012629853,"threshold_uncertainty_score":0.022529006},"labels":[],"label_agreement":null},{"id":"W4416271974","doi":"10.1007/s10664-025-10744-9","title":"An efficient model maintenance approach for MLOps","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reuse; Pipeline (software); Inference; Data modeling; Time series; Computation; Statistical model; Maintenance actions","score_opus":0.01994939686437478,"score_gpt":0.29261028354318575,"score_spread":0.27266088667881094,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416271974","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0075956504,0.000105829786,0.9848049,0.00025039917,0.000049299604,0.00014985062,0.00033216612,0.00555905,0.0011528048],"genre_scores_gemma":[0.12418984,0.00012944361,0.8704103,0.00015830355,0.00005285672,0.00020005835,0.0014268635,0.00079919427,0.0026331404],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99850756,0.00033801806,0.00013693006,0.00024433064,0.00066333986,0.00010978973],"domain_scores_gemma":[0.99391985,0.0017666792,0.0003835611,0.0024993666,0.0012999423,0.00013063058],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015194562,0.00094325654,0.00074873725,0.0018899307,0.0009801834,0.0020429941,0.0026035232,0.0008174766,0.005655235],"category_scores_gemma":[0.011766803,0.0007893746,0.0013005424,0.0016229523,0.00046806404,0.0035120358,0.0027399985,0.0019189424,0.0015319437],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021129381,0.00027557407,0.0034789,0.0003257819,0.00018683547,0.00036387955,0.00035810357,0.07113196,0.013782161,0.021719376,0.02065325,0.86751294],"study_design_scores_gemma":[0.00005460845,0.00008633141,0.0007361273,0.00004144068,0.00010919549,0.00025155293,0.0001947311,0.9351414,0.010143769,0.041752294,0.011464192,0.000024493595],"about_ca_topic_score_codex":0.0053168028,"about_ca_topic_score_gemma":0.012908298,"teacher_disagreement_score":0.005655235,"about_ca_system_score_codex":0.00096552353,"about_ca_system_score_gemma":0.0022914764,"threshold_uncertainty_score":0.018918633},"labels":[],"label_agreement":null},{"id":"W4417025955","doi":"10.1007/s10664-025-10779-y","title":"Immutable in principle, upgradeable by design: exploratory study of smart contract upgradeability","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Upgrade; Empirical research; Smart contract; Vulnerability (computing); Exploratory research; Exploit","score_opus":0.01826608134195462,"score_gpt":0.2740175608584428,"score_spread":0.2557514795164882,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417025955","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.993723,0.000035569123,0.003086903,0.00018176783,0.0000019671338,0.000026688916,0.000028877503,0.000010724552,0.0029044172],"genre_scores_gemma":[0.99889743,0.000018601151,0.0006809793,0.000012420259,0.0000013972899,0.000010178069,0.000036227408,0.0000046236396,0.00033801445],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99713373,0.001675434,0.00011703912,0.0002499453,0.0005997843,0.00022392711],"domain_scores_gemma":[0.8480573,0.1261557,0.009675765,0.010945101,0.003793528,0.0013726525],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010004078,0.00021196011,0.00026343265,0.0011486635,0.0012470263,0.0012702794,0.0010698396,0.0008196388,0.006065235],"category_scores_gemma":[0.091039576,0.0002743453,0.00023621261,0.0014881953,0.0036998405,0.005780035,0.001467219,0.0019417568,0.00027096312],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019237959,0.004638073,0.41197,0.000496936,0.00018576316,0.0012835712,0.028223902,0.024224931,0.003901835,0.393639,0.0022730646,0.12723917],"study_design_scores_gemma":[0.00048533423,0.002923931,0.29991442,0.00040664696,0.00018875419,0.0010434799,0.039108347,0.23820229,0.0076441146,0.39317703,0.016774543,0.00013110036],"about_ca_topic_score_codex":0.004190368,"about_ca_topic_score_gemma":0.0043650297,"teacher_disagreement_score":0.010004078,"about_ca_system_score_codex":0.0014455719,"about_ca_system_score_gemma":0.001730307,"threshold_uncertainty_score":0.05290723},"labels":[],"label_agreement":null},{"id":"W4417026560","doi":"10.1007/s10664-025-10772-5","title":"Automatically Detecting Checked-In Secrets in Android Apps: How Far Are We?","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds de recherche du Québec","keywords":"Android (operating system); Obfuscation; Cloud computing; Mobile device; Empirical research; Humanoid robot","score_opus":0.01545296850898984,"score_gpt":0.2740933592147346,"score_spread":0.25864039070574474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417026560","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.82874954,0.013515641,0.12165928,0.020036543,0.000838954,0.0002940644,0.0018023269,0.0012508284,0.01185278],"genre_scores_gemma":[0.9658237,0.0023028483,0.028438734,0.00095591,0.00026477262,0.00005683748,0.00085537235,0.00011420464,0.0011876202],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99275076,0.002993693,0.00048768663,0.0009439368,0.0024074346,0.00041656685],"domain_scores_gemma":[0.90133387,0.067774095,0.007847729,0.008785387,0.013060036,0.0011989027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00913162,0.0013357893,0.0010195144,0.002866845,0.00071497855,0.0042860806,0.00154811,0.002148022,0.0021898064],"category_scores_gemma":[0.0934786,0.0005259635,0.00080203085,0.0012965291,0.0017655351,0.00884126,0.0015175013,0.0027592713,0.0014139405],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032262353,0.00091537234,0.5078562,0.0010365635,0.0004601763,0.000285128,0.0013004619,0.0061129928,0.005854783,0.012833737,0.012531228,0.4504907],"study_design_scores_gemma":[0.00017497718,0.0012697495,0.44243237,0.0034524815,0.0010050318,0.002781696,0.00968604,0.3498609,0.021004058,0.124087155,0.04392988,0.0003157147],"about_ca_topic_score_codex":0.0035565693,"about_ca_topic_score_gemma":0.006740196,"teacher_disagreement_score":0.00913162,"about_ca_system_score_codex":0.0006700905,"about_ca_system_score_gemma":0.0018886063,"threshold_uncertainty_score":0.048293233},"labels":[],"label_agreement":null},{"id":"W4417026595","doi":"10.1007/s10664-025-10740-z","title":"Empirical studies of parameter efficient methods for large language models of code and knowledge transfer to R","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Okanagan University College; University of British Columbia, Okanagan Campus; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Automatic summarization; Empirical research; Code (set theory); Knowledge transfer; Language model; Resource (disambiguation)","score_opus":0.06950578563380058,"score_gpt":0.4078640021553274,"score_spread":0.33835821652152687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417026595","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28169933,0.0030173957,0.70471054,0.00292279,0.00006496153,0.00024326482,0.00044106023,0.0017438531,0.005156758],"genre_scores_gemma":[0.8778983,0.0011416062,0.11568182,0.00036194795,0.00013675336,0.0004900671,0.000815784,0.0011370161,0.0023366776],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.9496571,0.044985343,0.00082536385,0.0024034663,0.0016551723,0.0004734385],"domain_scores_gemma":[0.22624509,0.74220455,0.0071019,0.02062295,0.0030432688,0.00078231166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05642704,0.0017488376,0.0014692149,0.0029883548,0.0007910633,0.0031170866,0.0036012512,0.002415266,0.0047101784],"category_scores_gemma":[0.385297,0.0009453646,0.0016780372,0.0036754953,0.0029793398,0.011657476,0.0029857275,0.006062008,0.0013226924],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002477117,0.0019356097,0.049579706,0.0010637252,0.0017220274,0.00014895048,0.005335679,0.44873524,0.0028077431,0.15400125,0.008949274,0.32324362],"study_design_scores_gemma":[0.00031225142,0.00035416568,0.010207723,0.00016045506,0.00023286926,0.00013632704,0.00075303676,0.86896074,0.0018690195,0.11428139,0.0026446532,0.00008742771],"about_ca_topic_score_codex":0.005951645,"about_ca_topic_score_gemma":0.0045754514,"teacher_disagreement_score":0.05642704,"about_ca_system_score_codex":0.003247368,"about_ca_system_score_gemma":0.0021938027,"threshold_uncertainty_score":0.29841828},"labels":[],"label_agreement":null},{"id":"W4417031931","doi":"10.1007/s10664-025-10765-4","title":"Maintaining shared understanding of non-functional requirements in small companies using continuous software engineering","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; Microsoft (Canada); SST Wireless (Canada)","funders":"Science Foundation Ireland","keywords":"Maintainability; Rework; Personal software process; Social software engineering; Software development; Software Engineering Process Group; Context (archaeology); Software peer review; Software requirements; Software","score_opus":0.07692965972545587,"score_gpt":0.3050413437707852,"score_spread":0.2281116840453293,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417031931","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.982669,0.000029177236,0.0134578645,0.00043616391,0.0000050666613,0.00004324421,0.000008405687,0.000060898965,0.0032901943],"genre_scores_gemma":[0.9934464,0.000015191016,0.0059704715,0.00004730478,0.0000019765025,0.000024271678,0.000024359799,0.000025553303,0.00044446744],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9701681,0.017995197,0.0011428626,0.0017937927,0.007722802,0.0011771516],"domain_scores_gemma":[0.75388813,0.17212754,0.018823221,0.030454282,0.020671105,0.004035826],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02505369,0.00032861903,0.0003646146,0.0014233257,0.00166125,0.0055136466,0.0019298127,0.0015499989,0.00126319],"category_scores_gemma":[0.14614013,0.00051618076,0.000321699,0.00087180425,0.0023027225,0.0067990934,0.004725345,0.0021139083,0.00024032895],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051193,0.0019423804,0.21141465,0.00030023145,0.0001777973,0.0011722703,0.41176948,0.008329963,0.030731075,0.016733188,0.0018650392,0.3150519],"study_design_scores_gemma":[0.00018816932,0.0034701088,0.38474214,0.0007398338,0.00027466976,0.0018922623,0.36899313,0.13854699,0.026637413,0.047858175,0.026274553,0.00038267398],"about_ca_topic_score_codex":0.0050198254,"about_ca_topic_score_gemma":0.0062651625,"teacher_disagreement_score":0.02505369,"about_ca_system_score_codex":0.0026821308,"about_ca_system_score_gemma":0.005605168,"threshold_uncertainty_score":0.1324982},"labels":[],"label_agreement":null},{"id":"W7109084896","doi":"10.1007/s10664-025-10768-1","title":"Output format biases in the evaluation of large language models for code translation","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; Vector Institute","keywords":"Executable; Source code; Disk formatting; Code (set theory); Code review; Translation (biology); Empirical research; Software","score_opus":0.07234957541314808,"score_gpt":0.3707188625070804,"score_spread":0.29836928709393234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7109084896","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8468252,0.001633839,0.13430431,0.0019327302,0.00033322655,0.0003252197,0.0018938138,0.0042976458,0.008453984],"genre_scores_gemma":[0.9652106,0.00022131298,0.029844569,0.00033207235,0.00008796525,0.00016288679,0.0023049137,0.0010036937,0.00083190866],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9609644,0.030734723,0.0022993083,0.0019341254,0.0036167782,0.00045068658],"domain_scores_gemma":[0.58200186,0.38894907,0.005332486,0.0118328715,0.010684276,0.0011995554],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03710896,0.0014422827,0.0011425877,0.0022448876,0.0010562702,0.0047627543,0.0018145934,0.0025088727,0.003623045],"category_scores_gemma":[0.31157106,0.00071633206,0.0009893671,0.0024673468,0.002073056,0.0071293428,0.003557967,0.002905993,0.0012278149],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.011445871,0.002084312,0.08483738,0.0033373667,0.0014912729,0.00081946206,0.0073672715,0.31085327,0.017593995,0.03318355,0.018846728,0.50813943],"study_design_scores_gemma":[0.00092866394,0.0012454706,0.012000351,0.00038440563,0.0005370032,0.00033630387,0.0015364939,0.89983594,0.028812708,0.050049208,0.004175148,0.00015839397],"about_ca_topic_score_codex":0.0036910358,"about_ca_topic_score_gemma":0.0038109133,"teacher_disagreement_score":0.96289104,"about_ca_system_score_codex":0.0021573314,"about_ca_system_score_gemma":0.0020369978,"threshold_uncertainty_score":0.1962533},"labels":[],"label_agreement":null}]}