{"meta":{"query_hash":"c45847c6b3a0","filters":{"venue":"Evaluation and Assessment in Software Engineering"},"cohort_total":3,"direct_labels_cover":0,"predictions_cover":3,"exported":3,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/c45847c6b3a0","api":"https://metacan.xera.ac/api/v1/cohort?venue=Evaluation+and+Assessment+in+Software+Engineering"},"results":[{"id":"W3158412293","doi":"10.1145/3463274.3463342","title":"DABT: A Dependency-aware Bug Triaging Method","year":2021,"lang":"en","type":"article","venue":"Evaluation and Assessment in Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Software regression; Dependency (UML); Software bug; Security bug; Software; Blocking (statistics); Process (computing)","score_opus":0.03924068237700611,"score_gpt":0.3814679397277233,"score_spread":0.3422272573507172,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3158412293","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004825071,0.00018173047,0.9741072,0.00030860078,0.000086446074,0.00028719637,0.0004181748,0.01858227,0.0012033177],"genre_scores_gemma":[0.062381253,0.00017901885,0.9307789,0.00021418744,0.00006646167,0.00040541322,0.0014181694,0.001851789,0.0027047608],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974584,0.0005528953,0.00024837285,0.00058556267,0.0009639033,0.00019097058],"domain_scores_gemma":[0.9954485,0.002181303,0.00064791087,0.00057717744,0.00091921753,0.0002259169],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023228321,0.0018504716,0.0010415327,0.0045388583,0.0010469237,0.0013245852,0.0025306947,0.0011675851,0.005912902],"category_scores_gemma":[0.009691575,0.0010625655,0.002250424,0.0021052964,0.0009153255,0.0025041038,0.002184441,0.0020112032,0.0017767502],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002915968,0.00030882025,0.0071915705,0.0008232474,0.00018291708,0.0005883574,0.00050640217,0.03907168,0.020376727,0.015407263,0.046598606,0.8686529],"study_design_scores_gemma":[0.0003654498,0.00023722331,0.0020428556,0.00012771232,0.00024653116,0.0011232025,0.00022467822,0.89111346,0.020516891,0.03681283,0.047054306,0.00013475325],"about_ca_topic_score_codex":0.0055764713,"about_ca_topic_score_gemma":0.006891945,"teacher_disagreement_score":0.005912902,"about_ca_system_score_codex":0.0011491795,"about_ca_system_score_gemma":0.0043880013,"threshold_uncertainty_score":0.019780636},"labels":[],"label_agreement":null},{"id":"W3173944821","doi":"10.1145/3463274.3463343","title":"Assessing Developer Expertise from the Statistical Distribution of Programming Syntax Patterns","year":2021,"lang":"en","type":"article","venue":"Evaluation and Assessment in Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Task (project management); Syntax; Context (archaeology); Software engineering; Programming language; Data science; Artificial intelligence; Systems engineering; Engineering","score_opus":0.04264145128656933,"score_gpt":0.357883313491921,"score_spread":0.31524186220535166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173944821","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.75550115,0.0006291655,0.23651421,0.0003252994,0.000044971443,0.0001430695,0.0013431667,0.0008562924,0.004642776],"genre_scores_gemma":[0.97964305,0.00017604788,0.017988922,0.00005938741,0.000032439057,0.000100927115,0.0013497088,0.00011356991,0.00053586374],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98867315,0.0040612146,0.0009024495,0.0021792229,0.0038493513,0.00033460633],"domain_scores_gemma":[0.7794584,0.18113112,0.013933065,0.010855631,0.012651639,0.001970176],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015065602,0.0004979078,0.0006241573,0.006935353,0.0003688675,0.0015753087,0.0005832196,0.0011709041,0.0018434874],"category_scores_gemma":[0.14890976,0.00031658448,0.00045696818,0.0026548277,0.0010025717,0.0028228867,0.0016518224,0.0010519576,0.0010555524],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070430344,0.00022114662,0.7063872,0.00026678227,0.00035341273,0.00020099618,0.0011000755,0.02212478,0.009368991,0.0024783446,0.002964349,0.25382957],"study_design_scores_gemma":[0.00005580106,0.00069774315,0.6962022,0.00012014835,0.00011697637,0.0011864235,0.0008133033,0.26726508,0.009455976,0.02088236,0.00307769,0.00012625384],"about_ca_topic_score_codex":0.0013164079,"about_ca_topic_score_gemma":0.0017923896,"teacher_disagreement_score":0.015065602,"about_ca_system_score_codex":0.0004493401,"about_ca_system_score_gemma":0.0007459579,"threshold_uncertainty_score":0.079675436},"labels":[],"label_agreement":null},{"id":"W3176225318","doi":"10.1145/3463274.3463359","title":"Towards a More Structured Peer Review Process with Empirical Standards","year":2021,"lang":"en","type":"article","venue":"Evaluation and Assessment in Software Engineering","topic":"scientometrics and bibliometrics research","field":"Decision Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Dalhousie University","funders":"","keywords":"Process (computing); Computer science; Process management; Empirical research; Peer review; Business; Programming language; Political science","score_opus":0.4179059071787115,"score_gpt":0.6254396821821306,"score_spread":0.20753377500341913,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176225318","genre_codex":"methods","genre_gemma":"empirical","domain_codex":"methods","domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008466537,0.0039436733,0.90726477,0.048530594,0.0034703151,0.012567025,0.00016696173,0.0018777448,0.013712359],"genre_scores_gemma":[0.03794423,0.0014149768,0.9422894,0.005114207,0.0018758621,0.00845051,0.00028507988,0.00030199412,0.0023237152],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.18234229,0.61224675,0.04761062,0.028122041,0.12724149,0.0024369345],"domain_scores_gemma":[0.084445894,0.39266795,0.080729365,0.15688498,0.26790014,0.017371649],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.66490096,0.00245129,0.0039259195,0.019781187,0.008485331,0.030775122,0.0125985835,0.010609879,0.004877345],"category_scores_gemma":[0.7217215,0.0026711838,0.0029821002,0.011177309,0.02160466,0.023505172,0.024711717,0.01871015,0.0060409703],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036693658,0.001972328,0.008583817,0.01153228,0.0009439542,0.00067394326,0.039651856,0.0061592744,0.011672464,0.20055237,0.049841303,0.6680495],"study_design_scores_gemma":[0.001177796,0.0022512206,0.016164808,0.01925677,0.00035264078,0.0014451061,0.0159833,0.016353283,0.0075701466,0.44159198,0.4767262,0.001126702],"about_ca_topic_score_codex":0.0020757637,"about_ca_topic_score_gemma":0.0028683245,"teacher_disagreement_score":0.33509904,"about_ca_system_score_codex":0.014107258,"about_ca_system_score_gemma":0.08482743,"threshold_uncertainty_score":0.4132368},"labels":[],"label_agreement":null}]}