{"meta":{"query_hash":"bd4927a0c0a5","filters":{"venue":"Proceedings of the ACM on software engineering."},"cohort_total":36,"direct_labels_cover":0,"predictions_cover":36,"exported":36,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/bd4927a0c0a5","api":"https://metacan.xera.ac/api/v1/cohort?venue=Proceedings+of+the+ACM+on+software+engineering."},"results":[{"id":"W4394906543","doi":"10.1145/3660806","title":"An Empirical Study on Code Review Activity Prediction and Its Impact in Practice","year":2024,"lang":"en","type":"preprint","venue":"Proceedings of the ACM on software engineering.","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada); Queen's University","funders":"Mitacs","keywords":"Code (set theory); Computer science; Programming language","score_opus":0.10133249139045217,"score_gpt":0.44118744781838515,"score_spread":0.33985495642793295,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394906543","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9931004,0.0012374612,0.0025185074,0.00036146087,0.00003307489,0.00011891272,0.0012300286,0.00021624607,0.0011839735],"genre_scores_gemma":[0.9951351,0.0002650223,0.0021530027,0.000051186038,0.000035819256,0.00009275146,0.0018226663,0.000040736784,0.00040368282],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.97937816,0.009441298,0.0019449348,0.004036426,0.004567663,0.00063152367],"domain_scores_gemma":[0.514272,0.39707312,0.046840936,0.01119122,0.026393013,0.0042297207],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020706123,0.0006478004,0.00056089705,0.0034490693,0.00064405345,0.0022191235,0.0011951011,0.0011789443,0.0012053812],"category_scores_gemma":[0.19952227,0.00035503038,0.0006120332,0.0034946066,0.0010493177,0.0033526693,0.0011000376,0.0016256565,0.00096129754],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007159905,0.000677531,0.91784817,0.0008262271,0.0001977063,0.00022009505,0.0038948285,0.0031729117,0.0009901599,0.00023927963,0.0048974445,0.06631968],"study_design_scores_gemma":[0.00007425958,0.0015611496,0.9334898,0.0003260328,0.0001602689,0.00079868175,0.0034157098,0.049915187,0.002342713,0.0005415631,0.00727146,0.00010316699],"about_ca_topic_score_codex":0.004130891,"about_ca_topic_score_gemma":0.0039966456,"teacher_disagreement_score":0.020706123,"about_ca_system_score_codex":0.0012900616,"about_ca_system_score_gemma":0.0011489354,"threshold_uncertainty_score":0.10950571},"labels":[],"label_agreement":null},{"id":"W4400434361","doi":"10.1145/3715773","title":"An Adaptive Language-Agnostic Pruning Method for Greener Language Models for Code","year":2025,"lang":"en","type":"preprint","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Dalhousie University","funders":"Agencia Estatal de Investigación; Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada; Knut och Alice Wallenbergs Stiftelse","keywords":"Pruning; Computer science; Code (set theory); Artificial intelligence; Language model; Natural language processing; Programming language; Biology; Botany","score_opus":0.02915474167711867,"score_gpt":0.31616635034274054,"score_spread":0.28701160866562186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400434361","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02356315,0.00018692722,0.9668588,0.00013920067,0.000043022257,0.00007124526,0.00019660775,0.008108311,0.0008327282],"genre_scores_gemma":[0.24783222,0.0002408371,0.74167037,0.00035087785,0.000057169244,0.0002656317,0.0018483954,0.002052319,0.00568213],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99931073,0.00013638429,0.000040819566,0.00015996021,0.0002919869,0.00006009433],"domain_scores_gemma":[0.99846387,0.0007280737,0.00013902623,0.000291068,0.00033468675,0.00004329547],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00079750625,0.0008001346,0.00053734117,0.0010796604,0.00039378388,0.00077021134,0.0015387359,0.0006336298,0.0019760581],"category_scores_gemma":[0.0038108951,0.0003994316,0.0009833388,0.00063693535,0.00055192865,0.0014472075,0.0011911846,0.0013183923,0.0012535832],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003738739,0.00023456627,0.0038532645,0.00034539532,0.000117057476,0.00054117484,0.0005718721,0.14623164,0.09092842,0.016957905,0.012905958,0.72693884],"study_design_scores_gemma":[0.000021700365,0.00007223363,0.0004543065,0.00001708269,0.000024335008,0.00016673646,0.000053780062,0.96621424,0.021368982,0.006397401,0.0051913946,0.000017863144],"about_ca_topic_score_codex":0.0033419046,"about_ca_topic_score_gemma":0.009192916,"teacher_disagreement_score":0.0033419046,"about_ca_system_score_codex":0.00052977656,"about_ca_system_score_gemma":0.0013875167,"threshold_uncertainty_score":0.0066449046},"labels":[],"label_agreement":null},{"id":"W4400581753","doi":"10.1145/3660823","title":"Dependency-Induced Waste in Continuous Integration: An Empirical Study of Unused Dependencies in the npm Ecosystem","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Dependency (UML); Computer science; Context (archaeology); Reuse; JSON; Dependency theory (database theory); Code reuse; Resource (disambiguation); Dependency graph; Database; Software; Software engineering; Functional dependency; Relational database; Operating system; Engineering","score_opus":0.029745562251328024,"score_gpt":0.28998160021238495,"score_spread":0.26023603796105693,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400581753","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9954235,0.00043760048,0.0013746375,0.00022463594,0.000010665648,0.000035494602,0.0010858712,0.00009767838,0.0013099182],"genre_scores_gemma":[0.9881016,0.00043923347,0.0040228893,0.00018023749,0.000026732361,0.00010651256,0.006213979,0.00015512391,0.0007537696],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9890295,0.0032735667,0.0011278585,0.0018212011,0.0038077154,0.0009401053],"domain_scores_gemma":[0.84726363,0.088366844,0.03421008,0.012808301,0.012979541,0.004371649],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011716055,0.0006501609,0.0006282326,0.005044703,0.0014175612,0.0028533752,0.002490155,0.0012516541,0.0015672402],"category_scores_gemma":[0.088709,0.0006617882,0.0007560916,0.0086644655,0.0017857152,0.006918644,0.003775374,0.0025943716,0.00082960894],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019988825,0.00050731545,0.9575647,0.00027509604,0.00012157528,0.00054835127,0.004235549,0.001932406,0.0004901989,0.001034956,0.0034376446,0.02965235],"study_design_scores_gemma":[0.000030139714,0.00034457105,0.9494114,0.00028441552,0.000120257755,0.0012076317,0.012000858,0.02017725,0.0010056028,0.0026381628,0.012696265,0.000083334846],"about_ca_topic_score_codex":0.008248563,"about_ca_topic_score_gemma":0.008294825,"teacher_disagreement_score":0.011716055,"about_ca_system_score_codex":0.0014982012,"about_ca_system_score_gemma":0.0018353411,"threshold_uncertainty_score":0.061961174},"labels":[],"label_agreement":null},{"id":"W4400581925","doi":"10.1145/3660813","title":"Revealing Software Development Work Patterns with PR-Issue Graph Topologies","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Network topology; Computer science; Graph; Software development; Software; Software engineering; Theoretical computer science; Programming language; Operating system","score_opus":0.014321233758022994,"score_gpt":0.23332791828836394,"score_spread":0.21900668453034094,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400581925","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54024076,0.00089719024,0.4341388,0.0020073543,0.00006654034,0.00068969524,0.0036148457,0.002750587,0.015594171],"genre_scores_gemma":[0.74260086,0.00057677017,0.24983871,0.00010507762,0.000021845686,0.00046664482,0.0033443861,0.0004139714,0.0026317853],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99570256,0.002100476,0.00034330884,0.00065348257,0.0009875011,0.00021261531],"domain_scores_gemma":[0.95319545,0.034083117,0.004771584,0.004176746,0.0029716277,0.0008015468],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039060013,0.00049113913,0.00032603703,0.009624139,0.0014128542,0.0034472833,0.0010890975,0.0009801278,0.002034848],"category_scores_gemma":[0.029285898,0.0005487078,0.00061628834,0.008820276,0.0012279552,0.0073525296,0.0028362991,0.0010475747,0.0005609393],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047901,0.00045526173,0.2787185,0.0024390935,0.00029052387,0.0039727157,0.15341642,0.03542673,0.018053902,0.12364497,0.014754814,0.36834803],"study_design_scores_gemma":[0.0001144702,0.00034538718,0.14020117,0.001169973,0.00032335953,0.004348526,0.12295408,0.25894177,0.020589061,0.27950794,0.1711908,0.00031345268],"about_ca_topic_score_codex":0.003086518,"about_ca_topic_score_gemma":0.008144623,"teacher_disagreement_score":0.009624139,"about_ca_system_score_codex":0.0011728029,"about_ca_system_score_gemma":0.0013577886,"threshold_uncertainty_score":0.020657122},"labels":[],"label_agreement":null},{"id":"W4400581939","doi":"10.1145/3660812","title":"A Weak Supervision-Based Approach to Improve Chatbots for Code Repositories","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"AI in Service Interactions","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; Concordia University","funders":"","keywords":"Computer science; Code (set theory); Programming language; World Wide Web; Software engineering; Database","score_opus":0.012757721096490286,"score_gpt":0.24273440712088265,"score_spread":0.22997668602439236,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400581939","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18079951,0.0019436355,0.758029,0.00118289,0.0003271156,0.000926141,0.0013744297,0.05034285,0.005074461],"genre_scores_gemma":[0.61177045,0.0002887081,0.36872485,0.00083548354,0.00019936376,0.0008544615,0.0069546783,0.0012311676,0.009140789],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.993958,0.0024024413,0.00034543607,0.0019104466,0.0009784345,0.00040516985],"domain_scores_gemma":[0.9861813,0.006004495,0.0010581051,0.0023858414,0.0034667144,0.0009036016],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005951964,0.002241269,0.0018526537,0.002539399,0.0019250043,0.0013051983,0.0033082599,0.0020625354,0.0025971653],"category_scores_gemma":[0.019311441,0.0006568888,0.0011282337,0.0012675517,0.0015018228,0.004213799,0.0039050812,0.003041507,0.0022436546],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026331367,0.0022844558,0.027704919,0.0015515845,0.00021389648,0.0007028525,0.0042129643,0.037718464,0.06564067,0.0053673433,0.042099763,0.80986995],"study_design_scores_gemma":[0.00016054013,0.0008997137,0.0066412543,0.00013012384,0.00013404193,0.00033399882,0.00083569577,0.9467472,0.021723624,0.00613516,0.016176224,0.00008236658],"about_ca_topic_score_codex":0.013371117,"about_ca_topic_score_gemma":0.025752228,"teacher_disagreement_score":0.013371117,"about_ca_system_score_codex":0.0016111609,"about_ca_system_score_gemma":0.0036993097,"threshold_uncertainty_score":0.03147739},"labels":[],"label_agreement":null},{"id":"W4400582230","doi":"10.1145/3660810","title":"ClarifyGPT: A Framework for Enhancing LLM-Based Code Generation via Requirements Clarification","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"National Natural Science Foundation of China","keywords":"Consistency (knowledge bases); Computer science; Fidelity; Code (set theory); Natural language generation; Natural language; Artificial intelligence; Programming language","score_opus":0.03864211965604995,"score_gpt":0.2956780983054859,"score_spread":0.25703597864943595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400582230","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0033059258,0.00013131529,0.9238024,0.0003486057,0.000058803038,0.00056601514,0.00043907098,0.06945139,0.0018964295],"genre_scores_gemma":[0.036929715,0.00013779594,0.95135415,0.00037520428,0.000026763377,0.0006471363,0.0019025154,0.006736129,0.0018905431],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9939719,0.0024908965,0.0005322197,0.0010028702,0.0016720585,0.00033001867],"domain_scores_gemma":[0.9860563,0.0077466588,0.0012070166,0.0031370781,0.0013988545,0.00045412185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006709022,0.0025953315,0.00080138835,0.0023397529,0.0008374864,0.0020880944,0.0043767933,0.0024295985,0.008730297],"category_scores_gemma":[0.030963881,0.0017389818,0.0024571528,0.0009517119,0.0021650079,0.0038420695,0.004982915,0.004555638,0.004555242],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008521608,0.0010017274,0.0049293824,0.0030065512,0.0002532481,0.0016401716,0.004990538,0.07796579,0.06808623,0.07060592,0.08247399,0.6841943],"study_design_scores_gemma":[0.00043106094,0.0005191974,0.001237891,0.00045035666,0.000120660676,0.0011115681,0.0005765189,0.71627766,0.061271332,0.05199287,0.16573587,0.00027497933],"about_ca_topic_score_codex":0.005099506,"about_ca_topic_score_gemma":0.007589884,"teacher_disagreement_score":0.008730297,"about_ca_system_score_codex":0.0014769667,"about_ca_system_score_gemma":0.0039221975,"threshold_uncertainty_score":0.035481155},"labels":[],"label_agreement":null},{"id":"W4400582353","doi":"10.1145/3660809","title":"Mining Action Rules for Defect Reduction Planning","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Commit; Computer science; Counterfactual thinking; Reduction (mathematics); Precision and recall; Code (set theory); Action (physics); Recall; Compiler; Software; Baseline (sea); Machine learning; Software bug; Artificial intelligence; Software engineering; Programming language; Database; Set (abstract data type)","score_opus":0.03604877384191616,"score_gpt":0.29302717810681783,"score_spread":0.25697840426490165,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400582353","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.116558895,0.0010209694,0.8617915,0.0014327086,0.00011318466,0.0006972683,0.0035132961,0.01190358,0.0029685507],"genre_scores_gemma":[0.51082784,0.00036824172,0.47935903,0.00034230965,0.000038686194,0.0005673634,0.006789538,0.00038841105,0.001318645],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99678504,0.00082464906,0.00029560697,0.0008252639,0.0010762104,0.0001932515],"domain_scores_gemma":[0.9869556,0.009423043,0.0010989328,0.0009850197,0.0013317445,0.00020571308],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024896304,0.0017791917,0.00095891056,0.0038234573,0.0007189459,0.0013245814,0.0018124526,0.0013567694,0.00186069],"category_scores_gemma":[0.0148372585,0.000628332,0.0020836003,0.0014467494,0.00089189596,0.0015581672,0.0011349677,0.0016161189,0.00071035617],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004092722,0.00073523924,0.04425475,0.0011454223,0.0003545457,0.0018752434,0.0008452783,0.37752467,0.013810488,0.009574661,0.010250832,0.5392195],"study_design_scores_gemma":[0.000050226983,0.00012675238,0.0022259147,0.000089882946,0.00011422155,0.00025834487,0.00019572588,0.97296745,0.007828868,0.012385721,0.003722615,0.00003426463],"about_ca_topic_score_codex":0.010321472,"about_ca_topic_score_gemma":0.017946163,"teacher_disagreement_score":0.010321472,"about_ca_system_score_codex":0.0012241732,"about_ca_system_score_gemma":0.003765405,"threshold_uncertainty_score":0.020522773},"labels":[],"label_agreement":null},{"id":"W4400582376","doi":"10.1145/3660807","title":"Do Large Language Models Pay Similar Attention Like Human Programmers When Generating Code?","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Programmer; Interpretability; Computer science; Code (set theory); Programming language; Code generation; Artificial intelligence; Computer security","score_opus":0.01965023609466952,"score_gpt":0.26783801853326755,"score_spread":0.24818778243859804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400582376","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.70167685,0.0010546176,0.26437843,0.0054175216,0.00021407778,0.00017451786,0.0002560351,0.0065613403,0.020266566],"genre_scores_gemma":[0.96657914,0.00021170574,0.028446086,0.0015172684,0.00005028462,0.000067607485,0.00022581591,0.00097282726,0.0019292184],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99364907,0.0028781225,0.00019017086,0.0014672027,0.001347884,0.00046767507],"domain_scores_gemma":[0.9576057,0.027626192,0.0038080856,0.0071689407,0.0025850409,0.0012059471],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005873315,0.00068798196,0.0006284129,0.0010121072,0.000537013,0.002671429,0.0012312421,0.0015363065,0.0027466267],"category_scores_gemma":[0.07296127,0.00070400804,0.0005386031,0.0006570037,0.0017723308,0.0056012277,0.001913059,0.0017322255,0.0012328412],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015269577,0.00050549547,0.22723801,0.001074846,0.0004929402,0.0014335606,0.041294463,0.01489269,0.11579852,0.02094825,0.015584166,0.5592101],"study_design_scores_gemma":[0.00046317242,0.0016478384,0.2861018,0.00048916374,0.0006633696,0.004270142,0.023148201,0.35037914,0.072976165,0.14630745,0.1130025,0.0005509808],"about_ca_topic_score_codex":0.0029032642,"about_ca_topic_score_gemma":0.0038196777,"teacher_disagreement_score":0.005873315,"about_ca_system_score_codex":0.00075460045,"about_ca_system_score_gemma":0.00091580744,"threshold_uncertainty_score":0.03106141},"labels":[],"label_agreement":null},{"id":"W4400582478","doi":"10.1145/3660793","title":"Towards Better Graph Neural Network-Based Fault Localization through Enhanced Code Representation","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba; University of Alberta; Concordia University","funders":"","keywords":"Computer science; Debugging; Graph; Scalability; Inference; Software; Theoretical computer science; Leverage (statistics); Autoencoder; Artificial neural network; Software quality; Artificial intelligence; Data mining; Programming language; Software development","score_opus":0.019653216969188467,"score_gpt":0.27197463634407953,"score_spread":0.25232141937489105,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400582478","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030130295,0.0001958221,0.96422476,0.00028862545,0.000034285826,0.000045168512,0.00031056494,0.00358698,0.0011834014],"genre_scores_gemma":[0.54199314,0.00038226056,0.4504933,0.00030437042,0.00004348388,0.0002128326,0.0023403696,0.0004211197,0.0038091645],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996983,0.00006373087,0.000016357471,0.00010004172,0.00008833451,0.000033362652],"domain_scores_gemma":[0.9991837,0.0003149828,0.00011629817,0.00014577463,0.00020740379,0.000031888772],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00035818198,0.0009941454,0.00057570287,0.0015740021,0.0002441497,0.000738513,0.0014269209,0.00089097355,0.0018816788],"category_scores_gemma":[0.0026981542,0.0003532387,0.00071451237,0.0011138248,0.0005011826,0.0018087487,0.00079546607,0.0011954363,0.0005065162],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007591449,0.00007724834,0.0010232995,0.000089070745,0.000038793114,0.00007632711,0.000053414136,0.8097898,0.0062283427,0.0076752924,0.002531371,0.17234114],"study_design_scores_gemma":[0.0000024091394,0.0000072455578,0.00007042162,0.0000023681075,0.0000037078119,0.0000061636624,0.000003391206,0.9966917,0.00070041657,0.0022669842,0.00024329856,0.0000018665149],"about_ca_topic_score_codex":0.010944769,"about_ca_topic_score_gemma":0.01219986,"teacher_disagreement_score":0.010944769,"about_ca_system_score_codex":0.0010889341,"about_ca_system_score_gemma":0.00090454967,"threshold_uncertainty_score":0.021762133},"labels":[],"label_agreement":null},{"id":"W4400582740","doi":"10.1145/3643759","title":"Understanding and Detecting Annotation-Induced Faults of Static Analyzers","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Annotation; Computer science; Natural language processing; Artificial intelligence","score_opus":0.05424952230509408,"score_gpt":0.25652039502094065,"score_spread":0.20227087271584657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400582740","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6221396,0.00056736596,0.36708477,0.00040961747,0.000045028988,0.00017859196,0.000329288,0.007297831,0.0019479914],"genre_scores_gemma":[0.89629686,0.00016887565,0.10223923,0.0000771854,0.000021167927,0.00007430744,0.0003603895,0.00034183203,0.00042016816],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9916937,0.00253628,0.00065454084,0.0013431265,0.0031185264,0.00065381185],"domain_scores_gemma":[0.95423377,0.02915878,0.006627103,0.004783086,0.0048486446,0.00034872806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045727636,0.0011541982,0.00066720543,0.004685577,0.00061719154,0.0016917605,0.0014184066,0.0014098623,0.00075236603],"category_scores_gemma":[0.034991015,0.0006691461,0.00075708795,0.0016579948,0.0014223083,0.0037904705,0.0013207167,0.0010525219,0.00019902937],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008111766,0.00076092355,0.33311468,0.0011323874,0.0002583624,0.004228123,0.012503698,0.04721941,0.13472216,0.018392196,0.0031012595,0.44375566],"study_design_scores_gemma":[0.00008216942,0.00067356176,0.11254516,0.00045735372,0.0005780004,0.0031794435,0.0034139138,0.67184573,0.17401804,0.022553435,0.010456982,0.00019615686],"about_ca_topic_score_codex":0.0037077777,"about_ca_topic_score_gemma":0.004932108,"teacher_disagreement_score":0.004685577,"about_ca_system_score_codex":0.001179266,"about_ca_system_score_gemma":0.0016514387,"threshold_uncertainty_score":0.024183393},"labels":[],"label_agreement":null},{"id":"W4400582781","doi":"10.1145/3643775","title":"Improving the Learning of Code Review Successive Tasks with Cross-Task Knowledge Distillation","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Task (project management); Code (set theory); Distillation; Artificial intelligence; Machine learning; Programming language; Engineering; Chemistry; Chromatography","score_opus":0.013152326303382412,"score_gpt":0.2774240932908377,"score_spread":0.2642717669874553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400582781","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2945325,0.0061196527,0.65123683,0.0021974158,0.001041046,0.00086999574,0.0012806783,0.0360985,0.0066233682],"genre_scores_gemma":[0.76029,0.0007936453,0.21960555,0.0015196475,0.00034219163,0.00051749457,0.0051514828,0.00091879175,0.010861157],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955831,0.0015128047,0.00027555777,0.001565814,0.00075423275,0.0003084509],"domain_scores_gemma":[0.9782507,0.011806264,0.0015301799,0.00292512,0.004623816,0.0008640044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051596747,0.0024705927,0.0016729549,0.0019071148,0.00081820885,0.0017096024,0.0033676927,0.002824067,0.0023657097],"category_scores_gemma":[0.025589207,0.00072663365,0.001224848,0.0010752031,0.0009384231,0.0042698537,0.0031315458,0.0041490225,0.0018456471],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008641625,0.0010255239,0.0073400103,0.0008170148,0.0002262129,0.00026867623,0.0006103847,0.106452174,0.023082325,0.0014729665,0.021620931,0.8362196],"study_design_scores_gemma":[0.00009363012,0.00040024528,0.001384846,0.000045948058,0.0000744547,0.00011028905,0.00009769373,0.9759151,0.015545866,0.0028938218,0.0033904538,0.000047568825],"about_ca_topic_score_codex":0.0075472193,"about_ca_topic_score_gemma":0.012727662,"teacher_disagreement_score":0.0075472193,"about_ca_system_score_codex":0.0019552417,"about_ca_system_score_gemma":0.003066886,"threshold_uncertainty_score":0.027287304},"labels":[],"label_agreement":null},{"id":"W4400582900","doi":"10.1145/3643774","title":"AI-Assisted Code Authoring at Scale: Fine-Tuning, Deploying, and Mixed Methods Evaluation","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Code (set theory); Scale (ratio); Programming language; Geography; Cartography","score_opus":0.033052256068704086,"score_gpt":0.3336914157279161,"score_spread":0.300639159659212,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400582900","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5437647,0.0032712661,0.38323352,0.0024563,0.0007062853,0.0031225628,0.0023453876,0.045949865,0.01515006],"genre_scores_gemma":[0.53281695,0.00039613416,0.4541804,0.00075457914,0.000083789004,0.002540886,0.003033073,0.0038824757,0.0023117452],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98529196,0.009986321,0.0009595599,0.0014715168,0.0018485897,0.0004419814],"domain_scores_gemma":[0.86084145,0.11359395,0.0020227758,0.013958023,0.0077605206,0.0018232971],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02933411,0.0019187815,0.0007821088,0.001645065,0.0009150905,0.0021645639,0.003863279,0.0021455523,0.0029491577],"category_scores_gemma":[0.103697516,0.00089986477,0.0012223992,0.0010854851,0.0019176753,0.0029183433,0.0039117327,0.0034613619,0.0013994803],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0045553236,0.005040111,0.0458971,0.0039348523,0.0011117785,0.00064866175,0.007724445,0.24294372,0.022447033,0.016244328,0.034470096,0.6149826],"study_design_scores_gemma":[0.0012432722,0.0014389094,0.006654484,0.00037230502,0.00022238432,0.0001537621,0.0012140416,0.9453399,0.013294703,0.013540263,0.01639692,0.00012901807],"about_ca_topic_score_codex":0.0070797163,"about_ca_topic_score_gemma":0.008663188,"teacher_disagreement_score":0.02933411,"about_ca_system_score_codex":0.0016353606,"about_ca_system_score_gemma":0.002205478,"threshold_uncertainty_score":0.15513545},"labels":[],"label_agreement":null},{"id":"W4400582940","doi":"10.1145/3660786","title":"Demystifying Invariant Effectiveness for Securing Smart Contracts","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Invariant (physics); Computer security; Business; Computer science; Mathematics; Mathematical physics","score_opus":0.011554233350821282,"score_gpt":0.23141694031818386,"score_spread":0.21986270696736257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400582940","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.81649864,0.00061097625,0.16753474,0.00024614276,0.00003566038,0.00013596393,0.0007584468,0.011503427,0.0026760225],"genre_scores_gemma":[0.9525822,0.000104703504,0.046303604,0.000023457616,0.000007058572,0.000030197545,0.00051919394,0.0002044895,0.00022519387],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9958547,0.0010547033,0.00036076558,0.00056038005,0.0017634628,0.00040586383],"domain_scores_gemma":[0.9701201,0.018697642,0.003562542,0.0053625386,0.0018480716,0.00040919907],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043846425,0.0007089887,0.00046733094,0.004213819,0.00033167933,0.0015138959,0.0010428658,0.00057590567,0.0011052765],"category_scores_gemma":[0.028474644,0.00040943336,0.00056913303,0.001410002,0.0016664164,0.0029789642,0.0016426179,0.0011329763,0.00024665697],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010018849,0.00046813043,0.25779927,0.0006841977,0.0002837809,0.00068049954,0.0016483811,0.17941341,0.08766296,0.0182186,0.0027576205,0.4493813],"study_design_scores_gemma":[0.00004570802,0.00046334555,0.02756362,0.00006544257,0.000089440255,0.00042916092,0.00037226992,0.88121676,0.07695832,0.00998186,0.002732059,0.00008203759],"about_ca_topic_score_codex":0.002657651,"about_ca_topic_score_gemma":0.0037936359,"teacher_disagreement_score":0.0043846425,"about_ca_system_score_codex":0.00084288203,"about_ca_system_score_gemma":0.001352143,"threshold_uncertainty_score":0.023188472},"labels":[],"label_agreement":null},{"id":"W4400582991","doi":"10.1145/3643771","title":"RavenBuild: Context, Relevance, and Dependency Aware Build Outcome Prediction","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada); University of Waterloo","funders":"","keywords":"Relevance (law); Dependency (UML); Outcome (game theory); Context (archaeology); Computer science; Data science; Artificial intelligence; History; Political science; Mathematics","score_opus":0.009439507259860338,"score_gpt":0.22651776080869815,"score_spread":0.2170782535488378,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400582991","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41625404,0.0112907905,0.42615414,0.002047785,0.000699811,0.0007571366,0.025758853,0.106318146,0.010719327],"genre_scores_gemma":[0.7723374,0.00087572244,0.18968849,0.0004484473,0.00019756584,0.00039069567,0.029925885,0.0012095948,0.0049261954],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979724,0.00039303227,0.00012998954,0.0007831351,0.00052858924,0.00019281583],"domain_scores_gemma":[0.99630606,0.0018162136,0.00039310375,0.00062027154,0.00060654146,0.0002577607],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00233761,0.0031510452,0.0013615472,0.003886178,0.0005673679,0.0013036446,0.0022140183,0.0014795645,0.0022946226],"category_scores_gemma":[0.00963312,0.00065631187,0.0013310412,0.0018090033,0.00047228055,0.0023645968,0.0026853292,0.0025385392,0.0024646898],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011844503,0.0010411679,0.14637049,0.0009513492,0.0004885858,0.0010057238,0.0005459022,0.17958936,0.010462732,0.0020438377,0.060270503,0.596046],"study_design_scores_gemma":[0.000072765535,0.00032031554,0.015288595,0.00009471029,0.00013096347,0.00036242825,0.00011753078,0.95921636,0.0069761476,0.0053094467,0.0120436195,0.00006718086],"about_ca_topic_score_codex":0.0096813375,"about_ca_topic_score_gemma":0.024822254,"teacher_disagreement_score":0.0096813375,"about_ca_system_score_codex":0.0007128267,"about_ca_system_score_gemma":0.0012492613,"threshold_uncertainty_score":0.019249976},"labels":[],"label_agreement":null},{"id":"W4400583026","doi":"10.1145/3660780","title":"Do Words Have Power? Understanding and Fostering Civility in Code Review Discussion","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Legal Education and Practice Innovations","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Civility; Power (physics); Code (set theory); Sociology; Political science; Psychology; Computer science; Programming language; Law; Politics","score_opus":0.0743767049240331,"score_gpt":0.3572547891152694,"score_spread":0.28287808419123633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400583026","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5786625,0.009310001,0.034881774,0.20794815,0.001577964,0.00026701175,0.00004882222,0.00030803142,0.16699581],"genre_scores_gemma":[0.98736304,0.0014009607,0.0019132608,0.005373882,0.0003634663,0.00012544732,0.000020013938,0.00013900697,0.0033008803],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.80901045,0.15653343,0.0045436625,0.006608651,0.016256116,0.0070477724],"domain_scores_gemma":[0.5189419,0.38971597,0.045634285,0.012059797,0.020021725,0.013626336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.066027574,0.00089815794,0.0009118636,0.007814402,0.022809533,0.027620338,0.0034136015,0.008254616,0.004857671],"category_scores_gemma":[0.2938707,0.0013331388,0.00084490766,0.0032217966,0.06190302,0.04517166,0.029879173,0.009419537,0.00122374],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028349667,0.000020021933,0.0034946308,0.00014251813,0.000012789754,0.00060897414,0.9421662,0.000045543446,0.00028853,0.03865084,0.0025002286,0.012041435],"study_design_scores_gemma":[0.000035170062,0.00006597431,0.002969235,0.00081565086,0.000031652828,0.0008426611,0.78724384,0.00049066654,0.00054565707,0.08985628,0.11703151,0.00007174453],"about_ca_topic_score_codex":0.0051375683,"about_ca_topic_score_gemma":0.005652096,"teacher_disagreement_score":0.066027574,"about_ca_system_score_codex":0.013449858,"about_ca_system_score_gemma":0.018375972,"threshold_uncertainty_score":0.34919137},"labels":[],"label_agreement":null},{"id":"W4400583149","doi":"10.1145/3643731","title":"Characterizing Python Library Migrations","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Python (programming language); Computer science; Programming language","score_opus":0.04678773520401621,"score_gpt":0.2944841030300863,"score_spread":0.2476963678260701,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400583149","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8917039,0.0012427474,0.06233282,0.0007097376,0.00015880655,0.00049312675,0.013558848,0.018491933,0.011308062],"genre_scores_gemma":[0.7822109,0.0011972418,0.15194419,0.0008138504,0.000090818634,0.0014407866,0.047023203,0.0055460595,0.009732964],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99249345,0.001132235,0.00081068964,0.0016718252,0.003221868,0.00067006046],"domain_scores_gemma":[0.95884216,0.012065276,0.012292452,0.00755219,0.008251534,0.00099642],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037398003,0.00093353575,0.0004504528,0.0067852647,0.0017666657,0.0014372173,0.0016595687,0.0008087526,0.0014048651],"category_scores_gemma":[0.033258416,0.0007269117,0.00081447104,0.007560704,0.0014610974,0.0037168337,0.003577852,0.0016407102,0.0010284103],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073205214,0.0003569493,0.5347706,0.0025001736,0.00016652246,0.0023932252,0.016396271,0.008929322,0.035427622,0.00901311,0.037370138,0.35194394],"study_design_scores_gemma":[0.00006160542,0.000353856,0.6374365,0.0008199576,0.00021492172,0.0040014856,0.0068917987,0.060666543,0.061642945,0.008505732,0.21903318,0.00037152713],"about_ca_topic_score_codex":0.009194119,"about_ca_topic_score_gemma":0.019390445,"teacher_disagreement_score":0.009194119,"about_ca_system_score_codex":0.0022642778,"about_ca_system_score_gemma":0.0028767895,"threshold_uncertainty_score":0.019778192},"labels":[],"label_agreement":null},{"id":"W4411449683","doi":"10.1145/3729343","title":"VLATest: Testing and Evaluating Vision-Language-Action Models for Robotic Manipulation","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Robustness (evolution); Artificial intelligence; Software deployment; Machine learning; Generative grammar; Human–computer interaction; Software engineering","score_opus":0.04418603467234083,"score_gpt":0.3291355321729803,"score_spread":0.28494949750063947,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411449683","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6558537,0.00097640714,0.31557435,0.0006797121,0.00034323326,0.0012077261,0.001987693,0.016430315,0.006946956],"genre_scores_gemma":[0.83139485,0.0002398218,0.16296664,0.0002470521,0.000026491367,0.0006390127,0.0024863163,0.00062987837,0.0013699959],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983253,0.00057344924,0.0001157777,0.00040698075,0.00044464954,0.00013387788],"domain_scores_gemma":[0.9915685,0.0064569917,0.0004107166,0.00083907763,0.000481562,0.00024310288],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032024148,0.0016748578,0.0006235217,0.000867647,0.0005213654,0.0009809466,0.0034144383,0.0021770254,0.0030363582],"category_scores_gemma":[0.013307004,0.0007530313,0.0013651451,0.0003395785,0.0015721455,0.0019075904,0.0019317155,0.0021470238,0.0006799424],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073347037,0.00085010973,0.0057362136,0.00061672984,0.00020568751,0.00019154913,0.00026245785,0.8998533,0.012455232,0.0050714584,0.0040458185,0.069977894],"study_design_scores_gemma":[0.000049612237,0.00028203265,0.0005378685,0.000019749219,0.000014181999,0.000034756824,0.00002699867,0.9930728,0.004365382,0.0010002931,0.0005819615,0.0000142949475],"about_ca_topic_score_codex":0.01230621,"about_ca_topic_score_gemma":0.012434086,"teacher_disagreement_score":0.01230621,"about_ca_system_score_codex":0.0019022903,"about_ca_system_score_gemma":0.0016691722,"threshold_uncertainty_score":0.024469137},"labels":[],"label_agreement":null},{"id":"W4411449687","doi":"10.1145/3715730","title":"Towards Diverse Program Transformations for Program Simplification","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Concordia University","funders":"Engineering and Physical Sciences Research Council; Natural Sciences and Engineering Research Council of Canada","keywords":"Code refactoring; Program comprehension; Computer science; Source lines of code; Program transformation; Maintainability; Program slicing; Programming language; Heuristics; Software engineering; Set (abstract data type); Code (set theory); Static program analysis; Program analysis; Software; Source code; Software maintenance; Software quality; Software system; Software development; Operating system","score_opus":0.01895505048035933,"score_gpt":0.29468973969045453,"score_spread":0.2757346892100952,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411449687","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07891822,0.0007444045,0.8396897,0.00067746977,0.000076830656,0.0005430279,0.0019205789,0.07463872,0.002791066],"genre_scores_gemma":[0.16293325,0.0005369023,0.8156297,0.00043749553,0.00004907389,0.000443658,0.011663845,0.0066437284,0.0016623401],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9915176,0.0024091462,0.0007811349,0.0020944222,0.0028097387,0.00038799507],"domain_scores_gemma":[0.9787411,0.009060334,0.0018319228,0.006939915,0.0031377694,0.00028895529],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004671357,0.0020068905,0.0011746656,0.0049992283,0.0009833543,0.0019444107,0.0025175272,0.001193202,0.0018254946],"category_scores_gemma":[0.031859547,0.0012907973,0.0029241976,0.003388688,0.0015068981,0.0036079888,0.0038227183,0.0030333325,0.0020383922],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059569813,0.00079588726,0.033832572,0.0019375829,0.0003955423,0.00092092477,0.002834933,0.04990559,0.08773897,0.015261948,0.029888513,0.7758919],"study_design_scores_gemma":[0.00032165158,0.0007008758,0.014466686,0.00050207373,0.00046942977,0.0019399015,0.0010244648,0.6992407,0.13642214,0.04941817,0.095317505,0.00017639733],"about_ca_topic_score_codex":0.0024790931,"about_ca_topic_score_gemma":0.0061790803,"teacher_disagreement_score":0.0049992283,"about_ca_system_score_codex":0.0010069477,"about_ca_system_score_gemma":0.0026858693,"threshold_uncertainty_score":0.024704754},"labels":[],"label_agreement":null},{"id":"W4411449706","doi":"10.1145/3715738","title":"Code Change Intention, Development Artifact, and History Vulnerability: Putting Them Together for Vulnerability Fix Detection by LLM","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Manitoba","funders":"","keywords":"Computer science; Vulnerability (computing); Artifact (error); Commit; Context (archaeology); Leverage (statistics); Computer security; Vulnerability assessment; Data science; Artificial intelligence; Database; Psychology","score_opus":0.038219735064336666,"score_gpt":0.2594014123250999,"score_spread":0.2211816772607632,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411449706","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1830833,0.0029068182,0.70104325,0.0018807795,0.00026293017,0.00047720643,0.005141542,0.101660326,0.003543777],"genre_scores_gemma":[0.5136851,0.00039597557,0.4736831,0.00052343484,0.00005220999,0.00025833905,0.008393742,0.0009807955,0.002027283],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99719703,0.00085618434,0.00026203576,0.00087985065,0.0006101724,0.00019469908],"domain_scores_gemma":[0.9921863,0.0051772282,0.000526986,0.0010179378,0.000851364,0.00024019321],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026221853,0.0020513244,0.00065315387,0.0030863832,0.0005843059,0.0015358954,0.0020473455,0.0016244845,0.002722497],"category_scores_gemma":[0.012460486,0.0007147497,0.0017654208,0.0010426735,0.000717176,0.0035313678,0.0024375673,0.003296476,0.001783106],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006594124,0.0007115697,0.036223132,0.001098205,0.00026742942,0.0007885565,0.0013485575,0.078948855,0.032424204,0.0037419589,0.016732262,0.8270559],"study_design_scores_gemma":[0.000053135318,0.00015599193,0.0041927085,0.00007858465,0.00009801596,0.00025201682,0.00032006032,0.9633134,0.017162412,0.007351804,0.006948345,0.00007357218],"about_ca_topic_score_codex":0.013034832,"about_ca_topic_score_gemma":0.027008727,"teacher_disagreement_score":0.013034832,"about_ca_system_score_codex":0.0015424576,"about_ca_system_score_gemma":0.0025147721,"threshold_uncertainty_score":0.025917888},"labels":[],"label_agreement":null},{"id":"W4411449748","doi":"10.1145/3715735","title":"Hallucination Detection in Large Language Models with Metamorphic Relations","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Topic Modeling","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Calgary","funders":"","keywords":"Recall; Computer science; Margin (machine learning); Cognitive psychology; Psychology; Artificial intelligence; Machine learning","score_opus":0.00800568433846788,"score_gpt":0.20890885471749607,"score_spread":0.2009031703790282,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411449748","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21072254,0.0029204183,0.7291935,0.0013826889,0.00016596388,0.00027100442,0.0020285307,0.050792318,0.002523038],"genre_scores_gemma":[0.8225754,0.00040258074,0.17017597,0.0009175448,0.000077481665,0.00014293403,0.003407765,0.000859108,0.0014413692],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9957956,0.0016460613,0.00034602688,0.0010344423,0.0010014945,0.0001762769],"domain_scores_gemma":[0.9839357,0.011386541,0.0012417856,0.0021663848,0.0010183908,0.00025121096],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044403435,0.0016290031,0.00096973946,0.0016024695,0.0005243564,0.0019141111,0.0017742619,0.0015276129,0.0010413295],"category_scores_gemma":[0.027853578,0.00058618403,0.0014371332,0.0008889667,0.0010615648,0.0035927831,0.0030999796,0.0019480314,0.0009720192],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012957522,0.0004617297,0.038737785,0.0014161898,0.0010569214,0.002796693,0.0019496321,0.27758828,0.039259307,0.0066338177,0.021747962,0.60705596],"study_design_scores_gemma":[0.00004082602,0.000103991246,0.0016591571,0.000035742694,0.00006977429,0.00049041264,0.00022465237,0.9760089,0.00959558,0.009499172,0.0022344908,0.000037310532],"about_ca_topic_score_codex":0.0049680877,"about_ca_topic_score_gemma":0.007165468,"teacher_disagreement_score":0.0049680877,"about_ca_system_score_codex":0.0009038101,"about_ca_system_score_gemma":0.0010501721,"threshold_uncertainty_score":0.023483038},"labels":[],"label_agreement":null},{"id":"W4411449758","doi":"10.1145/3715734","title":"An Empirical Study on Release-Wise Refactoring Patterns","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Queen's University","funders":"","keywords":"Code refactoring; Computer science; Cohesion (chemistry); Quality (philosophy); Software deployment; Java; Code (set theory); Software evolution; Software engineering; Software; Programming language; Software system","score_opus":0.023231487738195986,"score_gpt":0.30415486897821586,"score_spread":0.28092338124001986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411449758","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99660265,0.000233511,0.0016347193,0.00013753073,0.0000067499404,0.0000699609,0.00025186728,0.0000383194,0.0010245097],"genre_scores_gemma":[0.9970715,0.0001469436,0.0017875796,0.000043634656,0.000007413488,0.00009229391,0.00038429344,0.000030411971,0.00043590713],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9879704,0.003681946,0.0016893588,0.0019486428,0.004063808,0.0006457414],"domain_scores_gemma":[0.6550581,0.20110068,0.09564765,0.011923021,0.03135919,0.004911321],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013191582,0.000385818,0.00032427444,0.0027133373,0.00071452634,0.0019519653,0.0011248356,0.0007648472,0.0013374736],"category_scores_gemma":[0.11947806,0.0005055139,0.00044012794,0.0034217234,0.0012370981,0.002800369,0.0012717552,0.0015443306,0.00041822914],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002143703,0.00024987536,0.9615933,0.00021691235,0.00007980589,0.00025676252,0.008212707,0.0005025575,0.0014705703,0.00035583443,0.0005192449,0.026328055],"study_design_scores_gemma":[0.0000158905,0.00033806317,0.9877395,0.000074194104,0.000030803985,0.00031340602,0.0065251254,0.0021522033,0.0006351253,0.00026513005,0.0018806283,0.000029827272],"about_ca_topic_score_codex":0.0024933196,"about_ca_topic_score_gemma":0.0036969848,"teacher_disagreement_score":0.013191582,"about_ca_system_score_codex":0.0011455609,"about_ca_system_score_gemma":0.0011218004,"threshold_uncertainty_score":0.069764555},"labels":[],"label_agreement":null},{"id":"W4411449759","doi":"10.1145/3715736","title":"One-for-All Does Not Work! Enhancing Vulnerability Detection by Mixture-of-Experts (MoE)","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Manitoba","funders":"","keywords":"Vulnerability (computing); Computer science; Task (project management); Artificial intelligence; Baseline (sea); Deep learning; Machine learning; Code (set theory); Computer security; Engineering; Biology","score_opus":0.008965197819905181,"score_gpt":0.24284575727169933,"score_spread":0.23388055945179415,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411449759","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.086546294,0.009282838,0.8327342,0.005954359,0.0010490662,0.00031320986,0.0022792642,0.05105182,0.010788998],"genre_scores_gemma":[0.526357,0.0024955154,0.44086093,0.005691069,0.00035056158,0.00025782268,0.0063377423,0.0028315757,0.014817716],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.995644,0.0011401688,0.00024382144,0.0014312487,0.0010189015,0.0005218375],"domain_scores_gemma":[0.99482083,0.0020488582,0.0003503031,0.0016626786,0.000766433,0.00035081228],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005310628,0.003987368,0.0023652045,0.0027693403,0.0009540166,0.0025124014,0.0032847903,0.003991501,0.0045119748],"category_scores_gemma":[0.013659122,0.001232172,0.003011136,0.0012458379,0.0014887467,0.009675007,0.0051804013,0.0053859805,0.0057078665],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009883525,0.0006475639,0.021451877,0.00077311683,0.001016319,0.00059035607,0.00036712454,0.0931206,0.015458069,0.008363274,0.07641772,0.7808056],"study_design_scores_gemma":[0.00008788864,0.00040561048,0.0027071624,0.00023014235,0.00024045749,0.0012734765,0.0002185164,0.90688676,0.020958865,0.03650761,0.030328048,0.00015553525],"about_ca_topic_score_codex":0.004856099,"about_ca_topic_score_gemma":0.010030233,"teacher_disagreement_score":0.005310628,"about_ca_system_score_codex":0.0012488324,"about_ca_system_score_gemma":0.0018461937,"threshold_uncertainty_score":0.02808559},"labels":[],"label_agreement":null},{"id":"W4411449762","doi":"10.1145/3729346","title":"CAShift: Benchmarking Log-Based Cloud Attack Detection under Normality Shift","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Cloud computing; Computer science; Normality; Benchmarking; Data mining; Construct (python library); Paradigm shift; Anomaly detection; Statistics; Mathematics; Operating system","score_opus":0.010564947580602788,"score_gpt":0.23294926959823797,"score_spread":0.2223843220176352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411449762","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8364463,0.0066995258,0.05632481,0.0018671753,0.0022056932,0.0011791085,0.031063521,0.055995867,0.008217947],"genre_scores_gemma":[0.8864237,0.00086091086,0.045213066,0.0005502438,0.00020527733,0.00026829034,0.06394218,0.00050531415,0.0020309389],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99689317,0.00047665535,0.00034495344,0.0009801558,0.0008828715,0.00042211488],"domain_scores_gemma":[0.9964618,0.0009592849,0.00038164612,0.0008567582,0.00095256243,0.0003879798],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002634072,0.0026651782,0.0011590447,0.0033194653,0.00088812347,0.0016155344,0.0030310005,0.0015424143,0.0012054984],"category_scores_gemma":[0.007795872,0.00040040127,0.0011119504,0.0023299472,0.0011507794,0.0025477696,0.0019139205,0.0019459429,0.0013249512],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004222825,0.0039528683,0.12597914,0.0026613474,0.0011899456,0.0010532863,0.00050075725,0.33758667,0.01862217,0.004351208,0.15328784,0.34659183],"study_design_scores_gemma":[0.00024782607,0.000881858,0.023975244,0.00006933073,0.00006466557,0.00046902368,0.00024063232,0.9503839,0.011423372,0.0016968213,0.010463392,0.00008401052],"about_ca_topic_score_codex":0.016928827,"about_ca_topic_score_gemma":0.018486047,"teacher_disagreement_score":0.016928827,"about_ca_system_score_codex":0.00177346,"about_ca_system_score_gemma":0.0018885208,"threshold_uncertainty_score":0.03366059},"labels":[],"label_agreement":null},{"id":"W4411449776","doi":"10.1145/3715766","title":"Automated and Accurate Token Transfer Identification and Its Applications in Cryptocurrency Security","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Security token; Computer science; Computer security; Cryptocurrency; Exploit; Identification (biology); False positive paradox; Artificial intelligence","score_opus":0.006555163432256633,"score_gpt":0.23174639618428244,"score_spread":0.22519123275202582,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411449776","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06591935,0.0010913401,0.9028373,0.00062286673,0.000094299874,0.00017257532,0.00032461894,0.025955126,0.002982548],"genre_scores_gemma":[0.6369744,0.0006773482,0.35827827,0.00015964676,0.00005681165,0.00008223194,0.0006494731,0.0006087672,0.002513091],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9966461,0.0009598567,0.00021078721,0.0007373819,0.0012258638,0.00021996823],"domain_scores_gemma":[0.98958147,0.004295861,0.0015184812,0.0029736955,0.0013531268,0.00027735604],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002478441,0.00083022023,0.0008165525,0.0024959962,0.0008825408,0.0018658956,0.0013873521,0.0014568341,0.002359014],"category_scores_gemma":[0.010561367,0.00067799794,0.0003988188,0.002021705,0.0016181853,0.004350273,0.001732637,0.0015034497,0.0013188237],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005961123,0.00039132757,0.017543914,0.00036476157,0.00006853621,0.0004228468,0.0004526117,0.117666915,0.0454879,0.028897695,0.010191585,0.77791584],"study_design_scores_gemma":[0.000041124313,0.00012781832,0.0021406184,0.00004816414,0.000019135256,0.0003983897,0.000102621685,0.92571306,0.03983193,0.022490004,0.009018427,0.00006874723],"about_ca_topic_score_codex":0.0042142873,"about_ca_topic_score_gemma":0.0033320864,"teacher_disagreement_score":0.0042142873,"about_ca_system_score_codex":0.0011847154,"about_ca_system_score_gemma":0.002206261,"threshold_uncertainty_score":0.013107359},"labels":[],"label_agreement":null},{"id":"W4411449886","doi":"10.1145/3729353","title":"LookAhead: Preventing DeFi Attacks via Unveiling Adversarial Contracts","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Adversarial system; Computer science; Database transaction; Focus (optics); Software deployment; Computer security; Artificial intelligence; Code (set theory); Semantics (computer science); State (computer science); Machine learning; Database; Programming language; Software engineering","score_opus":0.005683394962144602,"score_gpt":0.21751598185418827,"score_spread":0.21183258689204368,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411449886","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24907748,0.0014410168,0.72327197,0.0014092481,0.000121681434,0.00045557992,0.00084265176,0.017467832,0.0059126047],"genre_scores_gemma":[0.888467,0.00029915292,0.10729336,0.00035170792,0.000049355644,0.000095317126,0.0009404108,0.00020467746,0.0022989672],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9981287,0.0005338275,0.00011663485,0.0003803953,0.0006399565,0.0002005748],"domain_scores_gemma":[0.9948428,0.0020491371,0.0011171505,0.0013193144,0.00043117275,0.00024039965],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002248029,0.0011148682,0.0010129778,0.0017286679,0.00065364735,0.0014795491,0.0014131536,0.0013594637,0.0016252479],"category_scores_gemma":[0.00898841,0.00042720468,0.0007353798,0.0006860171,0.0014442984,0.004329004,0.0025590884,0.0018987871,0.0007998877],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009790785,0.0005507726,0.06005504,0.00037666393,0.00018091641,0.00056044303,0.0005820195,0.33094734,0.027522327,0.025600461,0.014730071,0.5379148],"study_design_scores_gemma":[0.00003064099,0.00017848451,0.00211615,0.000031092157,0.000020909594,0.00029352066,0.000085592575,0.9677633,0.009076159,0.017371958,0.0030005432,0.00003167632],"about_ca_topic_score_codex":0.0017653467,"about_ca_topic_score_gemma":0.0026683423,"teacher_disagreement_score":0.002248029,"about_ca_system_score_codex":0.00067480345,"about_ca_system_score_gemma":0.0016014408,"threshold_uncertainty_score":0.011888862},"labels":[],"label_agreement":null},{"id":"W4411449952","doi":"10.1145/3715729","title":"An Empirical Study of Suppressed Static Analysis Warnings","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Spectrum analyzer; False positive paradox; Python (programming language); Computer science; Software; Static analysis; Scalability; Empirical research; False positives and false negatives; Code (set theory); Artificial intelligence; Programming language; Statistics; Telecommunications; Operating system; Mathematics","score_opus":0.0138911653291388,"score_gpt":0.29776158001213404,"score_spread":0.28387041468299523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411449952","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9968978,0.00018848245,0.0012592062,0.0001713766,0.000009463416,0.000046594963,0.0001485798,0.000074419135,0.0012041285],"genre_scores_gemma":[0.9976841,0.00012762762,0.0012596372,0.00008177973,0.000012436458,0.000077315184,0.0002744886,0.00004030961,0.00044215942],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9807842,0.008362874,0.002045238,0.0023499785,0.0056635486,0.00079416466],"domain_scores_gemma":[0.6027452,0.23665886,0.094995245,0.018203497,0.041905228,0.0054919543],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019658288,0.00053117843,0.00039125548,0.0034282645,0.0010288821,0.0016436818,0.0012885832,0.00088185864,0.0013904454],"category_scores_gemma":[0.18644525,0.00058365555,0.00031004156,0.0025687465,0.0019270842,0.003912123,0.0020740998,0.0021394193,0.00042015786],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036868837,0.00045437046,0.90534794,0.00045970382,0.00010193359,0.0006015427,0.038610827,0.000521938,0.0026438616,0.0006614327,0.0017128662,0.04851492],"study_design_scores_gemma":[0.00004510229,0.0011233039,0.9454249,0.00035198082,0.00007777489,0.0012527549,0.030822007,0.00671124,0.0025591368,0.00095570274,0.010582405,0.00009365995],"about_ca_topic_score_codex":0.0019505593,"about_ca_topic_score_gemma":0.0023237583,"teacher_disagreement_score":0.019658288,"about_ca_system_score_codex":0.0006959908,"about_ca_system_score_gemma":0.0012327883,"threshold_uncertainty_score":0.10396415},"labels":[],"label_agreement":null},{"id":"W4411450040","doi":"10.1145/3715741","title":"Understanding and Characterizing Mock Assertions in Unit Tests","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Assertion; Test (biology); Testability; Complement (music); Unit testing; Programming language; Control flow; Test case; Software engineering; Reliability engineering; Machine learning; Software","score_opus":0.05208794963468529,"score_gpt":0.26277533769608935,"score_spread":0.21068738806140405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411450040","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16103145,0.0007137136,0.8311719,0.00054147386,0.00006836399,0.00031042792,0.0002785719,0.0030717538,0.00281233],"genre_scores_gemma":[0.7063916,0.00031969798,0.29005092,0.0003338907,0.0000673085,0.00043195055,0.0006862753,0.0007155598,0.0010028699],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9722415,0.012721437,0.0027931128,0.002828043,0.008027589,0.0013882483],"domain_scores_gemma":[0.71898556,0.19552024,0.029523859,0.03528652,0.01867695,0.0020069538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015965616,0.0011710208,0.00083554396,0.004196637,0.0009501083,0.0046691876,0.0027490049,0.0027275132,0.0013389476],"category_scores_gemma":[0.17984629,0.0013761406,0.0010335029,0.001986195,0.005152537,0.012391835,0.0034992145,0.0020848766,0.00051997474],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001001232,0.0006571236,0.15441646,0.0018502233,0.0002765476,0.00582518,0.01834038,0.14265609,0.040878158,0.3354688,0.004641004,0.29398882],"study_design_scores_gemma":[0.00010663513,0.00075552345,0.015785739,0.0012781398,0.00027138396,0.003743777,0.0026332967,0.5179235,0.053189196,0.3707064,0.03330152,0.00030478725],"about_ca_topic_score_codex":0.0030503094,"about_ca_topic_score_gemma":0.002743011,"teacher_disagreement_score":0.015965616,"about_ca_system_score_codex":0.0015584194,"about_ca_system_score_gemma":0.002327923,"threshold_uncertainty_score":0.084435225},"labels":[],"label_agreement":null},{"id":"W4411450091","doi":"10.1145/3715724","title":"CKTyper: Enhancing Type Inference for Java Code Snippets by Leveraging Crowdsourcing Knowledge in Stack Overflow","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Basic and Applied Basic Research Foundation of Guangdong Province","keywords":"Snippet; Computer science; Crowdsourcing; Context (archaeology); Code (set theory); Inference; Set (abstract data type); Information retrieval; Type inference; Java; Source code; World Wide Web; Artificial intelligence; Programming language","score_opus":0.017235033584320957,"score_gpt":0.2818270580510308,"score_spread":0.2645920244667099,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411450091","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08997626,0.00166789,0.80845,0.002548062,0.0008141303,0.0016208582,0.020026205,0.06303087,0.011865732],"genre_scores_gemma":[0.33157668,0.00072578294,0.61446005,0.001685774,0.00041396834,0.0015193606,0.03289322,0.0043893754,0.012335844],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99311054,0.0014570398,0.00043497572,0.002185591,0.0024254338,0.0003864954],"domain_scores_gemma":[0.9862316,0.007161349,0.0011921304,0.0028031862,0.0021257738,0.00048598484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004473238,0.0027966776,0.0012958692,0.0086967535,0.0021100813,0.0021237785,0.0029533668,0.0026795364,0.0041013528],"category_scores_gemma":[0.03155835,0.0008197028,0.002352061,0.0034475978,0.0016158328,0.0055519072,0.0049837944,0.0028493935,0.0029160038],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015639989,0.0007611581,0.027923858,0.0028117786,0.00048161994,0.0030401805,0.005110279,0.05347438,0.04961157,0.014269745,0.08702515,0.75392634],"study_design_scores_gemma":[0.0002922559,0.0002604919,0.015737887,0.00044509314,0.00029029336,0.0010085675,0.0019749457,0.76394,0.05160142,0.058879565,0.105106294,0.00046321267],"about_ca_topic_score_codex":0.026329812,"about_ca_topic_score_gemma":0.05241246,"teacher_disagreement_score":0.026329812,"about_ca_system_score_codex":0.0020531046,"about_ca_system_score_gemma":0.0037701996,"threshold_uncertainty_score":0.052353144},"labels":[],"label_agreement":null},{"id":"W4411450197","doi":"10.1145/3729363","title":"An Empirical Study of Bugs in Data Visualization Libraries","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Data Visualization and Analytics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Hong Kong University of Science and Technology","keywords":"Computer science; Visualization; Data science; Information retrieval; Root cause; Empirical research; Software bug; Key (lock); Data mining; Software; Programming language; Computer security","score_opus":0.03286994685591208,"score_gpt":0.33228897766324655,"score_spread":0.29941903080733445,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411450197","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9902249,0.0010338847,0.0049735554,0.000459453,0.000030366034,0.00023299294,0.0007246211,0.0008679438,0.001452313],"genre_scores_gemma":[0.991015,0.00047169143,0.006120857,0.00022906622,0.00001974404,0.00019174935,0.001034719,0.00021124558,0.00070583273],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9771482,0.0075052334,0.0030924918,0.0025001285,0.008599296,0.0011547833],"domain_scores_gemma":[0.6959439,0.19879138,0.05381682,0.011633495,0.03641923,0.0033950857],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015015036,0.001041518,0.0006899103,0.0068975184,0.001267297,0.0024089164,0.0017414668,0.0013135988,0.0017963953],"category_scores_gemma":[0.18803136,0.0007779203,0.00065647415,0.004226132,0.0018140397,0.005135497,0.0028548138,0.0017778929,0.0005551756],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005974667,0.000832202,0.8071393,0.0018079473,0.00017932952,0.0016619097,0.029772464,0.0011545906,0.0035920436,0.0013579611,0.0066531026,0.14525166],"study_design_scores_gemma":[0.0001093992,0.0020520932,0.895957,0.0018905381,0.00039537728,0.0042213853,0.035024095,0.019715799,0.010280414,0.0022531538,0.027803654,0.0002971117],"about_ca_topic_score_codex":0.003884543,"about_ca_topic_score_gemma":0.00469175,"teacher_disagreement_score":0.015015036,"about_ca_system_score_codex":0.001796522,"about_ca_system_score_gemma":0.0020848736,"threshold_uncertainty_score":0.07940805},"labels":[],"label_agreement":null},{"id":"W4411450220","doi":"10.1145/3715779","title":"Protecting Privacy in Software Logs: What Should Be Anonymized?","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Identification (biology); Computer science; Information privacy; Personally identifiable information; Privacy policy; Information sensitivity; Data anonymization; Software; Privacy by Design; Data science; Identifier; Internet privacy; Computer security","score_opus":0.018043817695445535,"score_gpt":0.2513165258656156,"score_spread":0.23327270817017007,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411450220","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1462523,0.02576927,0.5130436,0.25827911,0.0020384255,0.0020557928,0.009764718,0.0031611733,0.039635632],"genre_scores_gemma":[0.72342294,0.021985037,0.21460433,0.02108455,0.0017477681,0.002243602,0.009352604,0.0010841655,0.004474954],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.8911175,0.06561345,0.011584444,0.006970143,0.021604193,0.0031101878],"domain_scores_gemma":[0.6355125,0.20830248,0.030720308,0.081871696,0.040931668,0.0026614103],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06805147,0.00097007776,0.0018140778,0.005517486,0.004466236,0.017955747,0.0034816668,0.0041707414,0.0019857225],"category_scores_gemma":[0.25490007,0.0009920477,0.001512507,0.008291818,0.009983061,0.036752503,0.0071221334,0.0060766726,0.001402994],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005956238,0.0003888538,0.08581156,0.0053352998,0.00032224998,0.0011108463,0.036579303,0.009216866,0.0058205654,0.30378234,0.051637914,0.49939847],"study_design_scores_gemma":[0.00011089991,0.00022424286,0.027951634,0.01232043,0.00035029397,0.0018568868,0.056691617,0.01798371,0.013604618,0.46480098,0.4037194,0.00038529397],"about_ca_topic_score_codex":0.005782754,"about_ca_topic_score_gemma":0.004517729,"teacher_disagreement_score":0.06805147,"about_ca_system_score_codex":0.004048408,"about_ca_system_score_gemma":0.014638109,"threshold_uncertainty_score":0.35989487},"labels":[],"label_agreement":null},{"id":"W4411450398","doi":"10.1145/3729383","title":"Automated Extraction and Analysis of Developer's Rationale in Open Source Software","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Université de Montréal","funders":"","keywords":"Computer science; Software engineering; Software; Open source; Generalization; Open source software; Data science; Artificial intelligence; Programming language","score_opus":0.015592076908858506,"score_gpt":0.27953756571677185,"score_spread":0.26394548880791335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411450398","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12657768,0.00067045074,0.8495694,0.0018354555,0.00012375254,0.0009961085,0.002986051,0.013092906,0.0041483273],"genre_scores_gemma":[0.2689251,0.00032568575,0.72207373,0.00018695078,0.00005216657,0.0003411942,0.005399452,0.00085066666,0.0018449313],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98792946,0.0049188524,0.0010640712,0.0010860099,0.004616786,0.0003847305],"domain_scores_gemma":[0.94104934,0.038157392,0.006433772,0.0052429046,0.00860747,0.00050914765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012047728,0.0012338902,0.0006590258,0.010568864,0.0016689316,0.0037571178,0.0015421441,0.0016270586,0.0014001096],"category_scores_gemma":[0.052872553,0.0011308952,0.0014092717,0.0033692087,0.0011086505,0.005441541,0.0030662783,0.0024517085,0.0008733144],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027110858,0.0005134546,0.042458232,0.0015187131,0.00019671253,0.002444146,0.010098983,0.020425035,0.046405468,0.035266623,0.017582603,0.82281893],"study_design_scores_gemma":[0.0001687259,0.0003478054,0.03390755,0.0011938318,0.0003219292,0.0017818331,0.006127339,0.6698895,0.09076859,0.10928236,0.08572871,0.00048184037],"about_ca_topic_score_codex":0.0043702824,"about_ca_topic_score_gemma":0.013126711,"teacher_disagreement_score":0.012047728,"about_ca_system_score_codex":0.0015387858,"about_ca_system_score_gemma":0.0046923556,"threshold_uncertainty_score":0.06371522},"labels":[],"label_agreement":null},{"id":"W4411523015","doi":"10.1145/3728876","title":"MoDitector: Module-Directed Testing for Autonomous Driving Systems","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Root cause; Computer science; Debugging; Reliability (semiconductor); Reliability engineering; Root cause analysis; Process (computing); Scenario testing; Embedded system; Engineering; Artificial intelligence","score_opus":0.016940612286937884,"score_gpt":0.23593435620730108,"score_spread":0.2189937439203632,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411523015","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18658833,0.0006387783,0.73604816,0.00037834977,0.00014233353,0.00067971175,0.0009830919,0.06771174,0.006829457],"genre_scores_gemma":[0.70559293,0.00018296816,0.28782985,0.00028091646,0.000025238036,0.00034923333,0.0012271835,0.0019197102,0.0025920623],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99887186,0.00029428606,0.00006594134,0.00018593624,0.000471233,0.00011069236],"domain_scores_gemma":[0.9969541,0.0017778024,0.00030649416,0.0004991258,0.00036630774,0.000096147785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001273779,0.001298076,0.00033401162,0.0009163119,0.00025423104,0.0005513009,0.002340639,0.0009545211,0.003346474],"category_scores_gemma":[0.005496521,0.00043549106,0.0006335419,0.00028773415,0.00083833875,0.001353927,0.0009855236,0.00094404275,0.0005986017],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086137943,0.00049857004,0.02393115,0.0010338032,0.00022341085,0.0015035381,0.0008412727,0.33204192,0.16534922,0.013717144,0.016129771,0.4438688],"study_design_scores_gemma":[0.00012396037,0.0007011437,0.0030369745,0.0000635082,0.00006045945,0.0006703407,0.00007748,0.85589176,0.119308576,0.0066024144,0.013395785,0.00006758596],"about_ca_topic_score_codex":0.0024848378,"about_ca_topic_score_gemma":0.0031547588,"teacher_disagreement_score":0.003346474,"about_ca_system_score_codex":0.0005487496,"about_ca_system_score_gemma":0.00085482275,"threshold_uncertainty_score":0.011195064},"labels":[],"label_agreement":null},{"id":"W4411523021","doi":"10.1145/3728919","title":"Assessing Scene Generation Techniques for Testing COLREGS-Compliance of Autonomous Surface Vehicles","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Maritime Navigation and Safety","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Set (abstract data type); Computer science; Logical analysis; Functional requirement; Operations research; Artificial intelligence; Engineering; Software engineering","score_opus":0.04127833120726755,"score_gpt":0.2794976332482037,"score_spread":0.23821930204093617,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411523021","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.75819874,0.0003691014,0.23395424,0.0001699779,0.000030111118,0.00061879796,0.00044464122,0.0039399476,0.0022743659],"genre_scores_gemma":[0.7940692,0.00014187125,0.2040523,0.000047001467,0.0000070928045,0.00020983534,0.0009780793,0.00020448484,0.00029015826],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9950825,0.0024494634,0.00038672276,0.00067432167,0.0011623971,0.00024458102],"domain_scores_gemma":[0.969354,0.02330057,0.002130251,0.0026494316,0.002186199,0.00037960295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040415293,0.0010473178,0.00033095223,0.001402133,0.00024946217,0.0005826653,0.0018377816,0.0010158569,0.0009421203],"category_scores_gemma":[0.026636839,0.00034693012,0.0006860799,0.0008708245,0.0005971987,0.001196101,0.0010353161,0.0005837573,0.0002531559],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013983172,0.0017097724,0.0448966,0.0011041957,0.00031199213,0.00056554907,0.0012333759,0.4893942,0.09030801,0.0031896057,0.0019024622,0.36398587],"study_design_scores_gemma":[0.00017081376,0.0016871726,0.009125575,0.0000538517,0.000104480205,0.00040627405,0.00034838292,0.9337009,0.05142971,0.0014446257,0.0014944427,0.000033751057],"about_ca_topic_score_codex":0.0022622936,"about_ca_topic_score_gemma":0.0028475,"teacher_disagreement_score":0.0040415293,"about_ca_system_score_codex":0.00064316497,"about_ca_system_score_gemma":0.00080823654,"threshold_uncertainty_score":0.021373868},"labels":[],"label_agreement":null},{"id":"W4411523084","doi":"10.1145/3728947","title":"The First Prompt Counts the Most! An Evaluation of Large Language Models on Iterative Example-Based Code Generation","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Code (set theory); Computer science; Benchmark (surveying); Iterative and incremental development; Code generation; Process (computing); Natural language generation; Natural language; Programming language; Software engineering; Artificial intelligence; Computer security; Geography","score_opus":0.03728653452721785,"score_gpt":0.29708271316430906,"score_spread":0.2597961786370912,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411523084","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6797331,0.0032200613,0.26637298,0.0026740246,0.00041219144,0.0008985984,0.0021349262,0.024596568,0.019957498],"genre_scores_gemma":[0.6968561,0.00065398787,0.2905954,0.00051459094,0.00004288229,0.0004022537,0.0056602657,0.002079155,0.0031953412],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9839138,0.008508175,0.0007676229,0.0014534666,0.004884723,0.0004721425],"domain_scores_gemma":[0.9319447,0.045480896,0.0021422768,0.010447636,0.008788329,0.0011961933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010639799,0.0013957868,0.00089741754,0.0013409805,0.0005595933,0.002291641,0.0024234543,0.0015581814,0.0037069446],"category_scores_gemma":[0.074627034,0.0004931991,0.0010555172,0.0010932882,0.0012826596,0.0041286475,0.002586553,0.0021700477,0.0014507172],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032868145,0.0022746332,0.024948789,0.0030347393,0.00048793328,0.00059938116,0.0027288178,0.20937644,0.03705904,0.016428515,0.031635907,0.6681389],"study_design_scores_gemma":[0.00048029586,0.0025031138,0.008158939,0.00045163042,0.0002104257,0.00043140783,0.001215738,0.8979132,0.046336547,0.010256997,0.031866904,0.00017473844],"about_ca_topic_score_codex":0.0044884337,"about_ca_topic_score_gemma":0.0053305365,"teacher_disagreement_score":0.010639799,"about_ca_system_score_codex":0.0014666169,"about_ca_system_score_gemma":0.00235127,"threshold_uncertainty_score":0.056269348},"labels":[],"label_agreement":null},{"id":"W4411523089","doi":"10.1145/3728880","title":"Preventing Disruption of System Backup against Ransomware Attacks","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Ransomware; Backup; Computer science; Computer security; Encryption; Malware; Operating system","score_opus":0.006109823960931409,"score_gpt":0.22825175233185982,"score_spread":0.22214192837092842,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411523089","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.827668,0.0016971511,0.14807789,0.00039303163,0.00021759381,0.00020674054,0.00025682012,0.016250795,0.0052320375],"genre_scores_gemma":[0.95817924,0.00018861161,0.039478593,0.0001447097,0.000031568532,0.000027718464,0.000362876,0.00014538501,0.001441324],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99813545,0.00033683435,0.000102968195,0.00039401508,0.00079849106,0.00023225805],"domain_scores_gemma":[0.9955662,0.0010650142,0.0010569398,0.0013377991,0.0007871494,0.00018696507],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011724667,0.0011358459,0.00077578716,0.0018915667,0.00066351105,0.0008306207,0.00086665415,0.0009640473,0.0007391809],"category_scores_gemma":[0.007263238,0.00027821763,0.0004302582,0.0004067138,0.0005536036,0.0018051715,0.0013329592,0.00079975545,0.0011047318],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007871025,0.0006230848,0.0738163,0.0003787925,0.0002191248,0.00091961917,0.0007925273,0.020769631,0.12583528,0.0018406315,0.010779203,0.76323867],"study_design_scores_gemma":[0.0000610797,0.0026674978,0.063845895,0.00015380922,0.00021929122,0.0050089387,0.0008576065,0.5714704,0.33546633,0.0029434196,0.017147971,0.00015779355],"about_ca_topic_score_codex":0.00075083255,"about_ca_topic_score_gemma":0.0013130158,"teacher_disagreement_score":0.0018915667,"about_ca_system_score_codex":0.0002713082,"about_ca_system_score_gemma":0.00048200638,"threshold_uncertainty_score":0.006200671},"labels":[],"label_agreement":null},{"id":"W4411523132","doi":"10.1145/3728931","title":"Understanding Practitioners’ Expectations on Clear Code Review Comments","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Engineering Research","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"CLARITY; Computer science; Constructive; Relevance (law); Set (abstract data type); Code (set theory); Process (computing); Data science; Code review; Programming language; Software; Software quality; Political science; Software development","score_opus":0.058257941025824454,"score_gpt":0.2989452171374843,"score_spread":0.24068727611165983,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411523132","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7414639,0.0048350366,0.1850963,0.028308269,0.0008119354,0.0027745187,0.0010313868,0.004377065,0.03130161],"genre_scores_gemma":[0.9110454,0.0013644992,0.07729882,0.0037580007,0.00025619654,0.0013740979,0.0010332373,0.0007037598,0.0031659126],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.7449617,0.14688882,0.021116758,0.011231133,0.07154385,0.004257803],"domain_scores_gemma":[0.15616548,0.5341199,0.06310776,0.020182382,0.21961108,0.0068133534],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.19054729,0.0008006697,0.0008658379,0.008477343,0.0024751602,0.008369639,0.0021982556,0.0032094927,0.0025632652],"category_scores_gemma":[0.66641456,0.00095256424,0.0008508506,0.0034167124,0.0030519264,0.010792756,0.0054544252,0.0033987323,0.0022070475],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014045611,0.00045554998,0.1496185,0.007530111,0.00023845574,0.0011584088,0.17607987,0.004122589,0.0352952,0.01355615,0.043158352,0.56738234],"study_design_scores_gemma":[0.0007760562,0.0033703805,0.27468917,0.016904246,0.0007961949,0.0042357896,0.18924725,0.07326375,0.05308099,0.0470354,0.33487543,0.0017253795],"about_ca_topic_score_codex":0.0038341694,"about_ca_topic_score_gemma":0.004547938,"teacher_disagreement_score":0.19054729,"about_ca_system_score_codex":0.007260213,"about_ca_system_score_gemma":0.010797898,"threshold_uncertainty_score":0.9981993},"labels":[],"label_agreement":null}]}