{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":3,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":3,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"c45847c6b3a0","filters":{"venue":"Evaluation and Assessment in Software Engineering"}},"results":[{"id":"W3158412293","doi":"10.1145/3463274.3463342","title":"DABT: A Dependency-aware Bug Triaging Method","year":2021,"lang":"en","type":"article","venue":"Evaluation and Assessment in Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Software regression; Dependency (UML); Software bug; Security bug; Software; Blocking (statistics); Process (computing)","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.03924068237700611,"gpt":0.3814679397277233,"spread":0.3422272573507172,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002322832,0.001850472,0.001041533,0.004538858,0.001046924,0.001324585,0.002530695,0.001167585,0.005912902],"category_scores_gemma":[0.009691575,0.001062565,0.002250424,0.002105296,0.0009153255,0.002504104,0.002184441,0.002011203,0.00177675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00114918,"about_ca_system_score_gemma":0.004388001,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005576471,"about_ca_topic_score_gemma":0.006891945,"domain_scores_codex":[0.9974584,0.0005528953,0.0002483728,0.0005855627,0.0009639033,0.0001909706],"domain_scores_gemma":[0.9954485,0.002181303,0.0006479109,0.0005771774,0.0009192175,0.0002259169],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002915968,0.0003088203,0.00719157,0.0008232474,0.0001829171,0.0005883574,0.0005064022,0.03907168,0.02037673,0.01540726,0.04659861,0.8686529],"study_design_scores_gemma":[0.0003654498,0.0002372233,0.002042856,0.0001277123,0.0002465312,0.001123202,0.0002246782,0.8911135,0.02051689,0.03681283,0.04705431,0.0001347533],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004825071,0.0001817305,0.9741072,0.0003086008,0.00008644607,0.0002871964,0.0004181748,0.01858227,0.001203318],"genre_scores_gemma":[0.06238125,0.0001790189,0.9307789,0.0002141874,0.00006646167,0.0004054132,0.001418169,0.001851789,0.002704761],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005912902,"threshold_uncertainty_score":0.01978064,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3173944821","doi":"10.1145/3463274.3463343","title":"Assessing Developer Expertise from the Statistical Distribution of Programming Syntax Patterns","year":2021,"lang":"en","type":"article","venue":"Evaluation and Assessment in Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Task (project management); Syntax; Context (archaeology); Software engineering; Programming language; Data science; Artificial intelligence; Systems engineering; Engineering","authors":[{"name":"Arghavan Moradi Dakhel","is_ca":true},{"name":"Michel C. Desmarais","is_ca":true},{"name":"Foutse Khomh","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04264145128656933,"gpt":0.357883313491921,"spread":0.3152418622053517,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0150656,0.0004979078,0.0006241573,0.006935353,0.0003688675,0.001575309,0.0005832196,0.001170904,0.001843487],"category_scores_gemma":[0.1489098,0.0003165845,0.0004569682,0.002654828,0.001002572,0.002822887,0.001651822,0.001051958,0.001055552],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004493401,"about_ca_system_score_gemma":0.0007459579,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001316408,"about_ca_topic_score_gemma":0.00179239,"domain_scores_codex":[0.9886732,0.004061215,0.0009024495,0.002179223,0.003849351,0.0003346063],"domain_scores_gemma":[0.7794584,0.1811311,0.01393307,0.01085563,0.01265164,0.001970176],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0007043034,0.0002211466,0.7063872,0.0002667823,0.0003534127,0.0002009962,0.001100076,0.02212478,0.009368991,0.002478345,0.002964349,0.2538296],"study_design_scores_gemma":[0.00005580106,0.0006977431,0.6962022,0.0001201483,0.0001169764,0.001186423,0.0008133033,0.2672651,0.009455976,0.02088236,0.00307769,0.0001262538],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7555012,0.0006291655,0.2365142,0.0003252994,0.00004497144,0.0001430695,0.001343167,0.0008562924,0.004642776],"genre_scores_gemma":[0.979643,0.0001760479,0.01798892,0.00005938741,0.00003243906,0.0001009271,0.001349709,0.0001135699,0.0005358637],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0150656,"threshold_uncertainty_score":0.07967544,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3176225318","doi":"10.1145/3463274.3463359","title":"Towards a More Structured Peer Review Process with Empirical Standards","year":2021,"lang":"en","type":"article","venue":"Evaluation and Assessment in Software Engineering","topic":"scientometrics and bibliometrics research","field":"Decision Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University; Dalhousie University","funders":"","keywords":"Process (computing); Computer science; Process management; Empirical research; Peer review; Business; Programming language; Political science","authors":[{"name":"Arham Arshad","is_ca":true},{"name":"Taher A. Ghaleb","is_ca":true},{"name":"Paul Ralph","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4179059071787115,"gpt":0.6254396821821306,"spread":0.2075337750034191,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.664901,0.00245129,0.00392592,0.01978119,0.008485331,0.03077512,0.01259858,0.01060988,0.004877345],"category_scores_gemma":[0.7217215,0.002671184,0.0029821,0.01117731,0.02160466,0.02350517,0.02471172,0.01871015,0.00604097],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01410726,"about_ca_system_score_gemma":0.08482743,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002075764,"about_ca_topic_score_gemma":0.002868325,"domain_scores_codex":[0.1823423,0.6122468,0.04761062,0.02812204,0.1272415,0.002436935],"domain_scores_gemma":[0.08444589,0.3926679,0.08072937,0.156885,0.2679001,0.01737165],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0003669366,0.001972328,0.008583817,0.01153228,0.0009439542,0.0006739433,0.03965186,0.006159274,0.01167246,0.2005524,0.0498413,0.6680495],"study_design_scores_gemma":[0.001177796,0.002251221,0.01616481,0.01925677,0.0003526408,0.001445106,0.0159833,0.01635328,0.007570147,0.441592,0.4767262,0.001126702],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008466537,0.003943673,0.9072648,0.04853059,0.003470315,0.01256702,0.0001669617,0.001877745,0.01371236],"genre_scores_gemma":[0.03794423,0.001414977,0.9422894,0.005114207,0.001875862,0.00845051,0.0002850799,0.0003019941,0.002323715],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.335099,"threshold_uncertainty_score":0.4132368,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}