{"meta":{"query_hash":"83ada4cbf0c3","filters":{"venue":"Studies In Educational Evaluation"},"cohort_total":18,"direct_labels_cover":0,"predictions_cover":18,"exported":18,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/83ada4cbf0c3","api":"https://metacan.xera.ac/api/v1/cohort?venue=Studies+In+Educational+Evaluation"},"results":[{"id":"W1838440795","doi":"10.1016/j.stueduc.2015.09.002","title":"Evaluating intake variables for a teacher education programme: Improving student success and process efficiency","year":2015,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Teacher Education and Leadership Studies","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of the Fraser Valley","funders":"","keywords":"Practicum; Psychology; Medical education; Focus group; Process (computing); Program evaluation; Mathematics education; Pedagogy; Medicine; Computer science; Political science; Sociology","score_opus":0.43336915743112503,"score_gpt":0.5880200743970577,"score_spread":0.1546509169659327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1838440795","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99155957,0.0001496151,0.005459988,0.00012339873,0.00002598629,0.0009166984,0.00025358342,0.00014328407,0.001367856],"genre_scores_gemma":[0.9741794,0.00022729866,0.02035488,0.00006555116,0.000030621242,0.0022573564,0.0007614035,0.00010739471,0.0020161779],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9767542,0.014139234,0.0022465438,0.0008687625,0.005423336,0.00056794286],"domain_scores_gemma":[0.92289865,0.054015215,0.009599442,0.00361466,0.007284225,0.0025878872],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022812428,0.0011456083,0.0021196625,0.0018464942,0.0013485337,0.0031481048,0.001369392,0.0014076051,0.0030839574],"category_scores_gemma":[0.09686782,0.00050617085,0.0019029578,0.0025560898,0.0007339064,0.002380538,0.0029263047,0.0016883799,0.0006925259],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.038795453,0.038664266,0.39646107,0.0019077391,0.001649521,0.00010981986,0.008295959,0.005792647,0.010780621,0.0013974647,0.0021538148,0.4939916],"study_design_scores_gemma":[0.0038437918,0.05520778,0.86123264,0.00077547226,0.0028369885,0.00009639277,0.00439524,0.021046037,0.04417512,0.002454042,0.003648822,0.00028775315],"about_ca_topic_score_codex":0.0017080668,"about_ca_topic_score_gemma":0.0021507447,"teacher_disagreement_score":0.022812428,"about_ca_system_score_codex":0.0015613188,"about_ca_system_score_gemma":0.0032124016,"threshold_uncertainty_score":0.120645106},"labels":[],"label_agreement":null},{"id":"W1976592601","doi":"10.1016/j.stueduc.2013.10.006","title":"Towards a culture of inquiry for data use in schools: Breaking down professional learning barriers through intentional interruption","year":2013,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":78,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Status quo; Context (archaeology); Interrupt; Psychology; Cognition; Professional development; Professional learning community; Social psychology; Pedagogy; Computer science; Political science","score_opus":0.6261356046479121,"score_gpt":0.6152237617204723,"score_spread":0.010911842927439741,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1976592601","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45567238,0.0014003091,0.3881127,0.1015556,0.0005802963,0.0014002925,0.00008948642,0.0011789318,0.050009985],"genre_scores_gemma":[0.9237673,0.00023405578,0.0697734,0.003569662,0.00005111349,0.00076849444,0.000028622262,0.00020507495,0.0016022612],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.6036176,0.33051986,0.01731797,0.01039383,0.032960657,0.005190037],"domain_scores_gemma":[0.3914138,0.40700424,0.033139307,0.08809414,0.060837265,0.019511249],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.23884258,0.0007802983,0.0011232953,0.0035502783,0.01183689,0.034365747,0.005737445,0.0061613834,0.0018814211],"category_scores_gemma":[0.35697973,0.0017615528,0.0011576291,0.0020122833,0.042016108,0.022263946,0.031105272,0.016943227,0.00073138415],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011888727,0.0006642709,0.02527545,0.00061179453,0.000084187755,0.00017743993,0.8474998,0.00031473016,0.002600784,0.060313456,0.0019930152,0.060346182],"study_design_scores_gemma":[0.00017114094,0.00095604575,0.019802565,0.0036135092,0.00022851772,0.0012352339,0.74529105,0.004737841,0.008387514,0.117403455,0.09783654,0.0003366186],"about_ca_topic_score_codex":0.0046820645,"about_ca_topic_score_gemma":0.0031213493,"teacher_disagreement_score":0.23884258,"about_ca_system_score_codex":0.008506613,"about_ca_system_score_gemma":0.033200342,"threshold_uncertainty_score":0.9386426},"labels":[],"label_agreement":null},{"id":"W1982497397","doi":"10.1016/s0191-491x(02)00014-7","title":"Matching the grade 8 TIMSS item pool to the Ontario curriculum","year":2002,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto; Lakehead University","funders":"","keywords":"Mathematics education; Curriculum; Matching (statistics); Secondary education; Psychology; Pedagogy; Mathematics; Statistics","score_opus":0.1656116291663467,"score_gpt":0.46858849052707463,"score_spread":0.30297686136072793,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982497397","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8653688,0.00028221757,0.008903453,0.0009755732,0.00026292226,0.010193389,0.01556348,0.00047175394,0.09797835],"genre_scores_gemma":[0.8571397,0.0004773014,0.023602365,0.00054757274,0.00007735432,0.019713577,0.024123287,0.00027405508,0.074044734],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99457586,0.0010798334,0.0008775836,0.0004309253,0.0022610568,0.00077460957],"domain_scores_gemma":[0.98238695,0.0018775178,0.0013449743,0.0016088217,0.011788719,0.0009930335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006091932,0.00045520285,0.00087595836,0.00277918,0.0016319698,0.0013194004,0.0010017052,0.0005862346,0.015324642],"category_scores_gemma":[0.02805678,0.00035272838,0.0011925163,0.0031187374,0.0006877443,0.00051318377,0.0016168247,0.00055141206,0.005923534],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020997983,0.0014681094,0.6317484,0.0006047713,0.00024538164,0.00022016288,0.0063142935,0.0016011576,0.0107468255,0.0025860693,0.0780332,0.26433188],"study_design_scores_gemma":[0.00015866513,0.00038370848,0.95599735,0.0001328983,0.00007836178,0.00004161184,0.001492595,0.000821356,0.0018499247,0.0003503727,0.038662747,0.00003048803],"about_ca_topic_score_codex":0.48082414,"about_ca_topic_score_gemma":0.7813034,"teacher_disagreement_score":0.9921031,"about_ca_system_score_codex":0.007896907,"about_ca_system_score_gemma":0.024597902,"threshold_uncertainty_score":0.9560509},"labels":[],"label_agreement":null},{"id":"W2054415360","doi":"10.1016/j.stueduc.2009.12.005","title":"Evaluation for learning: A cross-case analysis of evaluator strategies","year":2009,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Queen's University","funders":"","keywords":"Process (computing); Citizen journalism; Participatory evaluation; Reflection (computer programming); Program evaluation; Knowledge management; Computer science; Process management; Psychology; Sociology; Political science; Engineering","score_opus":0.5328970254040822,"score_gpt":0.6892803130688502,"score_spread":0.15638328766476794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2054415360","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9527079,0.0024272164,0.029621936,0.00040798867,0.000025754809,0.0015249925,0.0001421793,0.00006327751,0.013078674],"genre_scores_gemma":[0.98555124,0.0004081047,0.012458278,0.00006874457,0.0000050911804,0.00046527304,0.00011564356,0.000032770546,0.0008948744],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9202376,0.063063234,0.0043513114,0.0016262671,0.008702668,0.0020189928],"domain_scores_gemma":[0.5661487,0.38191277,0.009329598,0.008887236,0.0318068,0.0019149337],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07237728,0.0006741517,0.0007788126,0.008694193,0.0018672548,0.0044468404,0.0018394846,0.0016778823,0.004266267],"category_scores_gemma":[0.22874193,0.00031697378,0.0010484288,0.0033171126,0.0014968878,0.0066439384,0.0034919525,0.0009560552,0.00046955424],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0045742593,0.003982893,0.3122345,0.0030474083,0.0011962617,0.0018986167,0.11247794,0.0046221362,0.0043048034,0.025376568,0.0025289732,0.5237557],"study_design_scores_gemma":[0.0010906495,0.013331426,0.5261942,0.00706409,0.004737012,0.00645377,0.2626217,0.070883155,0.028905021,0.041864637,0.036197886,0.0006564897],"about_ca_topic_score_codex":0.0028600458,"about_ca_topic_score_gemma":0.003915768,"teacher_disagreement_score":0.07237728,"about_ca_system_score_codex":0.0050665685,"about_ca_system_score_gemma":0.0042618453,"threshold_uncertainty_score":0.3827722},"labels":[],"label_agreement":null},{"id":"W2072591409","doi":"10.1016/s0191-491x(00)00011-0","title":"The Brazilian National Evaluation System of Basic Education: Context, process, and impact","year":2000,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"School Choice and Performance","field":"Social Sciences","cited_by":23,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Context (archaeology); Process (computing); Process management; Mathematics education; Computer science; Psychology; Business; Geography","score_opus":0.08028795285571905,"score_gpt":0.49237274027401395,"score_spread":0.4120847874182949,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2072591409","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.59071845,0.029642379,0.029144207,0.05950306,0.0014190701,0.004488849,0.008014493,0.0007250669,0.27634436],"genre_scores_gemma":[0.9611255,0.002641059,0.024838915,0.0015944483,0.00017206237,0.001362812,0.0017734292,0.00012748502,0.006364251],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.90502805,0.05855149,0.008763349,0.0030717396,0.020747608,0.0038378164],"domain_scores_gemma":[0.7653102,0.10226599,0.012936095,0.013729062,0.09477784,0.010980867],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1265112,0.00054023037,0.0016262021,0.01043102,0.004089256,0.006270562,0.0018207966,0.0016677879,0.0030093747],"category_scores_gemma":[0.15320775,0.000543949,0.00087087514,0.011235123,0.00397084,0.002853333,0.0045088315,0.0014458335,0.00041147787],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007408187,0.0010691169,0.2833598,0.0049915924,0.0002800091,0.00020527169,0.020074418,0.002464334,0.0021562127,0.11871902,0.035958197,0.52998114],"study_design_scores_gemma":[0.0001452611,0.00071973575,0.79728496,0.0046004592,0.00042713832,0.00025416628,0.012734545,0.004137269,0.002217122,0.014330208,0.16295443,0.00019467025],"about_ca_topic_score_codex":0.1547793,"about_ca_topic_score_gemma":0.2098379,"teacher_disagreement_score":0.1547793,"about_ca_system_score_codex":0.03396073,"about_ca_system_score_gemma":0.11871337,"threshold_uncertainty_score":0.6690632},"labels":[],"label_agreement":null},{"id":"W2089211529","doi":"10.1016/j.stueduc.2007.04.002","title":"ANTIBULLYING PROGRAMS: A SURVEY OF EVALUATION ACTIVITIES IN PUBLIC SCHOOLS","year":2007,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Bullying, Victimization, and Aggression","field":"Psychology","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Social Sciences and Humanities Research Council of Canada; Ministère de l’Éducation, Gouvernement de l’Ontario","keywords":"Rigour; Psychological intervention; Program evaluation; Medical education; Poison control; Human factors and ergonomics; Suicide prevention; Medicine; Injury prevention; Psychology; Applied psychology; Nursing; Environmental health; Political science","score_opus":0.21528422940679026,"score_gpt":0.48333294453285675,"score_spread":0.2680487151260665,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2089211529","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99805486,0.0005858287,0.000076878125,0.00018739385,0.000006030611,0.00007765596,0.00015953326,0.000009534005,0.00084226124],"genre_scores_gemma":[0.9978709,0.0008214549,0.00022347098,0.00018621392,0.0000082078805,0.000090184156,0.00023506736,0.000009472727,0.0005548833],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99415827,0.0019359231,0.0008351754,0.0003742195,0.0015332432,0.001163179],"domain_scores_gemma":[0.974779,0.0048922542,0.00999663,0.00081817474,0.0049310806,0.0045827585],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005622582,0.00032847337,0.00060214085,0.004612078,0.0018080951,0.0014388231,0.0006764196,0.0010123608,0.0011777731],"category_scores_gemma":[0.01403124,0.0005533772,0.00050456455,0.0033862463,0.00087467226,0.0013027168,0.0019054435,0.001316322,0.00025727783],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000101532954,0.00079687865,0.97424906,0.000117536685,0.000035042434,0.000106821775,0.0054501817,0.000026938882,0.0004317019,0.00006609768,0.00062329695,0.017994817],"study_design_scores_gemma":[0.000005067406,0.00020667072,0.9883675,0.000060947834,0.000012681008,0.00008939294,0.010251761,0.00004649486,0.00012899913,0.000012934943,0.0008102614,0.000007337504],"about_ca_topic_score_codex":0.020466331,"about_ca_topic_score_gemma":0.041295156,"teacher_disagreement_score":0.020466331,"about_ca_system_score_codex":0.0025278786,"about_ca_system_score_gemma":0.0047747623,"threshold_uncertainty_score":0.040694416},"labels":[],"label_agreement":null},{"id":"W2522686050","doi":"10.1016/j.stueduc.2016.08.007","title":"Meta-analysis of faculty's teaching effectiveness: Student evaluation of teaching ratings and student learning are not related","year":2016,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":584,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mount Royal University","funders":"","keywords":"Psychology; Set (abstract data type); Meta-analysis; Student achievement; Sample (material); Mathematics education; Artifact (error); Correlation; Academic achievement; Computer science; Mathematics","score_opus":0.45477448836377427,"score_gpt":0.6028893736354598,"score_spread":0.14811488527168554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2522686050","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15649684,0.81872046,0.0116320625,0.0018928639,0.0044383556,0.001524181,0.0020941736,0.0003873671,0.0028137637],"genre_scores_gemma":[0.96146774,0.029092817,0.00475117,0.0014263041,0.00069108116,0.0007900942,0.00068141375,0.00021755895,0.0008817526],"study_design_codex":"meta_analysis","study_design_gemma":"meta_analysis","domain_scores_codex":[0.83660954,0.106706485,0.030482829,0.013199877,0.011249429,0.0017517766],"domain_scores_gemma":[0.7019669,0.23933363,0.02390463,0.02377444,0.008627466,0.0023930273],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.09887735,0.004661966,0.01960776,0.0058062333,0.001935917,0.0061967685,0.0045609735,0.0049941293,0.0039072447],"category_scores_gemma":[0.2658324,0.0028642032,0.07199819,0.0053607197,0.0031177816,0.0040190523,0.0034372725,0.0045874794,0.00055564434],"study_design_candidate":"meta_analysis","study_design_consensus":"meta_analysis","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037410164,0.0000207989,0.006799675,0.0120750265,0.97453916,0.000056524223,0.00013293604,0.00011043584,0.0001293513,0.00006449238,0.00020737371,0.0021233084],"study_design_scores_gemma":[0.002422841,0.0005270611,0.0073197912,0.0017000538,0.98671234,0.00008444957,0.00007854356,0.00018606878,0.00015833962,0.00029148316,0.00048801675,0.00003104724],"about_ca_topic_score_codex":0.009112501,"about_ca_topic_score_gemma":0.016407117,"teacher_disagreement_score":0.9011226,"about_ca_system_score_codex":0.0037121205,"about_ca_system_score_gemma":0.003121431,"threshold_uncertainty_score":0.52291965},"labels":[],"label_agreement":null},{"id":"W2782375796","doi":"10.1016/j.stueduc.2017.12.008","title":"Re-conceptualizing classroom assessment fairness: A systematic meta-ethnography of assessment literature and beyond","year":2018,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":82,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Conceptualization; Accountability; Construct (python library); Educational assessment; Psychology; Ethnography; Standards-based assessment; Process (computing); Pedagogy; Alternative assessment; Mathematics education; Sociology; Computer science; Political science","score_opus":0.18245862922193032,"score_gpt":0.5132713437504184,"score_spread":0.33081271452848804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2782375796","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32379025,0.53355265,0.11014269,0.017364152,0.0015529931,0.006101578,0.00087981997,0.00015620768,0.0064596264],"genre_scores_gemma":[0.8689556,0.05447753,0.06654858,0.0037876095,0.00019515507,0.0051614554,0.0003582157,0.00011946918,0.00039633585],"study_design_codex":"qualitative","study_design_gemma":"systematic_review","domain_scores_codex":[0.74442184,0.20750585,0.027224764,0.00859302,0.010349663,0.0019049065],"domain_scores_gemma":[0.26141024,0.6871106,0.018560624,0.019281922,0.012772307,0.00086427375],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.277291,0.0016118253,0.004838464,0.012877875,0.0028553717,0.009569612,0.0040061055,0.0021620586,0.0016451152],"category_scores_gemma":[0.45874414,0.0016026213,0.0036364826,0.008381753,0.0069407676,0.017215496,0.008853901,0.0051892106,0.00010388512],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072752335,0.00033388272,0.028040595,0.12765524,0.013514762,0.00041418595,0.4336383,0.0014811318,0.001736386,0.02693461,0.0032764212,0.36224693],"study_design_scores_gemma":[0.0005584494,0.0009179476,0.034198668,0.4212689,0.03022719,0.0009804715,0.38144717,0.003945926,0.004883492,0.066341914,0.05475406,0.00047586448],"about_ca_topic_score_codex":0.010198089,"about_ca_topic_score_gemma":0.019743057,"teacher_disagreement_score":0.722709,"about_ca_system_score_codex":0.0092114415,"about_ca_system_score_gemma":0.025475541,"threshold_uncertainty_score":0.89122885},"labels":[],"label_agreement":null},{"id":"W3128610517","doi":"10.1016/j.stueduc.2021.100977","title":"The long-term washback effects of the National Matriculation English Test on college English learning in China: Tertiary student perspectives","year":2021,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":31,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Matriculation; Active listening; Competence (human resources); Psychology; Test (biology); College English; Curriculum; Perception; Mathematics education; China; Medical education; Pedagogy; Medicine; Political science","score_opus":0.031240674061682346,"score_gpt":0.419786253443572,"score_spread":0.38854557938188966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3128610517","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9982646,0.00035499144,0.00001883549,0.0004816446,0.000014368723,0.000013837267,0.00003073175,0.0000022870556,0.00081861735],"genre_scores_gemma":[0.9995789,0.000080930906,0.000016335003,0.00006223969,0.000012572828,0.0000076219385,0.000030081681,8.14016e-7,0.00021038395],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.99647236,0.001296135,0.00026966757,0.00023770386,0.00096069946,0.0007635829],"domain_scores_gemma":[0.97766805,0.0072433725,0.003860228,0.00095970574,0.004833297,0.00543532],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009496037,0.00024421114,0.00071884005,0.0011130775,0.0014864964,0.001189498,0.0010148213,0.0009780076,0.00194772],"category_scores_gemma":[0.031687226,0.00013025544,0.00073730736,0.0009764917,0.0010759194,0.0010617089,0.0014470808,0.0011897513,0.00020472503],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022453002,0.0032216853,0.9122018,0.000090587484,0.00013663569,0.00043454798,0.0058540325,0.00022369935,0.00087918754,0.00031708955,0.00092382217,0.07347168],"study_design_scores_gemma":[0.000048326398,0.0022951558,0.9930386,0.00004275153,0.00010097034,0.000049111215,0.0032249033,0.00025318423,0.00044754625,0.000084966625,0.00039585735,0.000018655577],"about_ca_topic_score_codex":0.04799857,"about_ca_topic_score_gemma":0.058846485,"teacher_disagreement_score":0.04799857,"about_ca_system_score_codex":0.0037282226,"about_ca_system_score_gemma":0.005394956,"threshold_uncertainty_score":0.09543836},"labels":[],"label_agreement":null},{"id":"W3176398386","doi":"10.1016/j.stueduc.2021.101058","title":"The development and psychometric properties of an educational development impact questionnaire","year":2021,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Alberta Advanced Education; University of Alberta","funders":"","keywords":"Exploratory factor analysis; Context (archaeology); Psychology; Scholarship; Psychometrics; Medical education; Applied psychology; Mathematics education; Knowledge management; Computer science; Medicine; Developmental psychology","score_opus":0.3343641478523127,"score_gpt":0.5538289167297261,"score_spread":0.2194647688774134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176398386","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9268235,0.00061670627,0.049629394,0.0011807072,0.00021872071,0.005961272,0.0015816743,0.00038662856,0.013601352],"genre_scores_gemma":[0.922534,0.0004610147,0.06828327,0.0002658688,0.000059119295,0.006047794,0.0013017514,0.00006663323,0.0009805465],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9733478,0.013215705,0.0042001368,0.0007843532,0.0076795146,0.0007725439],"domain_scores_gemma":[0.889928,0.07327176,0.0076799295,0.004035634,0.022658223,0.0024264776],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.043181997,0.00048706628,0.0007027555,0.003720639,0.0006963392,0.0018190303,0.0010300777,0.00071119884,0.0017488329],"category_scores_gemma":[0.09779668,0.00041461503,0.0013760456,0.002872629,0.0009118128,0.001995589,0.0022613476,0.0013416371,0.0005299097],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046725536,0.0016380986,0.5691129,0.0005933691,0.00026889626,0.00013125426,0.0043902495,0.0024399557,0.0022002042,0.0028316858,0.0031942881,0.41273195],"study_design_scores_gemma":[0.00020982367,0.004352064,0.9527038,0.0005000644,0.00019569286,0.000364641,0.00416931,0.011199944,0.0036004086,0.0028604695,0.019687207,0.00015669633],"about_ca_topic_score_codex":0.0013022716,"about_ca_topic_score_gemma":0.0014580167,"teacher_disagreement_score":0.043181997,"about_ca_system_score_codex":0.0013031705,"about_ca_system_score_gemma":0.0027383205,"threshold_uncertainty_score":0.22837096},"labels":[],"label_agreement":null},{"id":"W4210694249","doi":"10.1016/j.stueduc.2022.101126","title":"Item wording effects in self-report measures and reading achievement: Does removing careless respondents help?","year":2022,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reading (process); Psychology; Mathematics education; Academic achievement; Achievement test; Standardized test; Linguistics","score_opus":0.4604167025598831,"score_gpt":0.5511538330204064,"score_spread":0.09073713046052334,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4210694249","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9271633,0.007903477,0.048782445,0.006741693,0.0012597849,0.001231604,0.0006427927,0.00028908395,0.0059858873],"genre_scores_gemma":[0.9501703,0.00105609,0.041562803,0.0027587283,0.0002712315,0.0012528241,0.00078101264,0.0001726229,0.0019742649],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.81026065,0.13629869,0.025513615,0.0087472135,0.017522693,0.0016571622],"domain_scores_gemma":[0.342504,0.5609976,0.028592745,0.046007343,0.01991794,0.0019803813],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.24155103,0.0012781048,0.002030071,0.0021498513,0.0021651797,0.002463259,0.0033217364,0.0035447667,0.0031294373],"category_scores_gemma":[0.4629652,0.0014592722,0.0045160353,0.0025324516,0.00386732,0.0063607586,0.002083638,0.003431642,0.00064702233],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028581694,0.0018518289,0.69569105,0.0019489516,0.002746594,0.00026478947,0.014272297,0.00052705436,0.0030957346,0.0035446687,0.005233288,0.26796553],"study_design_scores_gemma":[0.00065229635,0.0027394036,0.95886713,0.0025498886,0.003280166,0.0006680186,0.0045050383,0.002246629,0.009854341,0.0054942896,0.0089103645,0.00023248095],"about_ca_topic_score_codex":0.004975556,"about_ca_topic_score_gemma":0.0101468405,"teacher_disagreement_score":0.75844896,"about_ca_system_score_codex":0.0013949656,"about_ca_system_score_gemma":0.0027178596,"threshold_uncertainty_score":0.9353026},"labels":[],"label_agreement":null},{"id":"W4385393971","doi":"10.1016/j.stueduc.2023.101290","title":"Evaluating student engagement and experiential learning in global classrooms: A qualitative case study","year":2023,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Education and Critical Thinking Development","field":"Social Sciences","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University; University of Toronto","funders":"","keywords":"Experiential learning; Student engagement; Psychology; Interactivity; Pedagogy; Mathematics education; Experiential education; Educational technology; Qualitative research; Sociology; Computer science; Multimedia; Social science","score_opus":0.4136159316534047,"score_gpt":0.6442659677685991,"score_spread":0.2306500361151944,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385393971","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9854866,0.00024849948,0.0035865,0.0011901183,0.00004506347,0.0005132441,0.000054467135,0.000019713338,0.008855843],"genre_scores_gemma":[0.9950734,0.00017802074,0.0015273687,0.00024124481,0.000009918466,0.00031778964,0.000021207596,0.000015551956,0.0026153456],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9747541,0.019206844,0.00055395946,0.0010018304,0.0021663955,0.0023168847],"domain_scores_gemma":[0.94600934,0.041483663,0.0022915092,0.0013508537,0.004026296,0.004838276],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025079234,0.0006983465,0.001052707,0.0019723827,0.012376981,0.0070921425,0.002886346,0.002683048,0.0028334754],"category_scores_gemma":[0.037282772,0.000541829,0.00056845695,0.0018243402,0.009622476,0.003330367,0.009121965,0.0034374178,0.00037391606],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000107717446,0.0009362479,0.007332212,0.00022371506,0.000011247152,0.0010903121,0.968154,0.00012253865,0.0013593568,0.0025168248,0.0004919263,0.017653812],"study_design_scores_gemma":[0.000016963111,0.00030966246,0.0032525766,0.00018290954,0.000012024094,0.00030142712,0.9890219,0.00018184354,0.0012005875,0.0006974479,0.0048012375,0.000021399092],"about_ca_topic_score_codex":0.0057588047,"about_ca_topic_score_gemma":0.01306193,"teacher_disagreement_score":0.025079234,"about_ca_system_score_codex":0.0074757705,"about_ca_system_score_gemma":0.0077550667,"threshold_uncertainty_score":0.13263327},"labels":[],"label_agreement":null},{"id":"W4387762598","doi":"10.1016/j.stueduc.2023.101309","title":"Predicting reading comprehension performance based on student characteristics and item properties","year":2023,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Reading and Literacy Development","field":"Psychology","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centre for Advancing Health Outcomes; University of Alberta","funders":"","keywords":"Reading comprehension; Reading (process); Affect (linguistics); Test (biology); Psychology; Multilingualism; Comprehension; Mathematics education; Cognition; Computer science; Cognitive psychology; Linguistics; Pedagogy; Communication","score_opus":0.1548524743564766,"score_gpt":0.43847279524195376,"score_spread":0.2836203208854772,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387762598","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9991968,0.000025907488,0.00038343068,0.000016147744,0.000002067613,0.000007721726,0.00006841623,0.000017201763,0.00028242817],"genre_scores_gemma":[0.99902344,0.000021346108,0.0004248909,0.00000775893,0.000003783143,0.000010865117,0.00027295825,0.000010228085,0.0002246675],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99869967,0.00059769326,0.00018325151,0.00016619873,0.00024851167,0.00010469116],"domain_scores_gemma":[0.9447168,0.043135777,0.0051296046,0.0014745004,0.0034428437,0.0021004789],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032820578,0.00067754445,0.0004951226,0.0019591611,0.00025778462,0.001556087,0.0003933065,0.0011496729,0.0027348814],"category_scores_gemma":[0.032076698,0.00019970741,0.0007783767,0.0013953422,0.00029719132,0.0014864957,0.0006504689,0.00078735035,0.0012323363],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022999567,0.00026098784,0.9900137,0.0000111810305,0.00006664135,0.00003619784,0.00015649156,0.0003936042,0.0012143083,0.000021778469,0.00010892777,0.0074860607],"study_design_scores_gemma":[0.000025674437,0.0010781976,0.9871367,0.000014221565,0.00009657328,0.00012615148,0.00043014495,0.0083761215,0.00241516,0.00017811569,0.000110739755,0.000012227459],"about_ca_topic_score_codex":0.0012841638,"about_ca_topic_score_gemma":0.0015659664,"teacher_disagreement_score":0.0032820578,"about_ca_system_score_codex":0.0002509395,"about_ca_system_score_gemma":0.00043879412,"threshold_uncertainty_score":0.01735735},"labels":[],"label_agreement":null},{"id":"W4402569962","doi":"10.1016/j.stueduc.2024.101403","title":"Students’ perceptions of online peer feedback in process-oriented L2 writing: A qualitative inquiry","year":2024,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Peer feedback; Perception; Mathematics education; Peer evaluation; Process (computing); Qualitative research; Psychology; Computer science; Pedagogy; Higher education; Sociology; Political science","score_opus":0.23009569155715098,"score_gpt":0.6111867833582557,"score_spread":0.38109109180110473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402569962","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99555486,0.00012740419,0.0018251367,0.00053504726,0.00001548412,0.00013938153,0.00003334557,0.000013494559,0.0017558186],"genre_scores_gemma":[0.997503,0.00015934547,0.0010452936,0.00015878495,0.000006445446,0.0001971091,0.000018422546,0.00000848891,0.0009032363],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.98728186,0.009311726,0.00051884796,0.0005317061,0.001257487,0.0010985091],"domain_scores_gemma":[0.9780919,0.016617626,0.0013300873,0.00052749045,0.0020649438,0.0013679942],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015004843,0.00042001388,0.00064845855,0.001518649,0.0041360967,0.0043237824,0.0011274694,0.0018408947,0.0015380153],"category_scores_gemma":[0.024770908,0.00045521607,0.00055835006,0.0011898918,0.0050795167,0.003100005,0.0035449688,0.0021692428,0.00028316406],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047299836,0.0002121761,0.009103642,0.00019004566,0.000008766821,0.0006879005,0.9774415,0.00011543926,0.0019366825,0.00061252294,0.00017636764,0.009467658],"study_design_scores_gemma":[0.000012713867,0.00032145542,0.007881106,0.00018551844,0.000012008743,0.00041090098,0.982278,0.0004324356,0.001694788,0.0005163736,0.0062193633,0.00003531458],"about_ca_topic_score_codex":0.0019630503,"about_ca_topic_score_gemma":0.0021174448,"teacher_disagreement_score":0.015004843,"about_ca_system_score_codex":0.0022368247,"about_ca_system_score_gemma":0.0029214711,"threshold_uncertainty_score":0.07935417},"labels":[],"label_agreement":null},{"id":"W4404020856","doi":"10.1016/j.stueduc.2024.101412","title":"Predicting the Mathematics Literacy of Resilient Students from High‐performing Economies: A Machine Learning Approach","year":2024,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Mathematics education; Literacy; Computer science; Sociology; Pedagogy; Mathematics","score_opus":0.050506115022949764,"score_gpt":0.4089772244966975,"score_spread":0.3584711094737477,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404020856","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99313015,0.00023299381,0.004519769,0.0004662746,0.00001941266,0.00006172779,0.00040228496,0.000046836947,0.0011206017],"genre_scores_gemma":[0.99512345,0.00014325716,0.0036289752,0.000051357358,0.000009235517,0.00009483479,0.0005479526,0.000004628558,0.0003963776],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99939466,0.00026942405,0.000052496904,0.00011409547,0.00008330146,0.000085974425],"domain_scores_gemma":[0.99817383,0.0009318782,0.00030444827,0.00012610952,0.00027011207,0.00019352887],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017165053,0.00060163,0.00048401605,0.0020047582,0.00067515625,0.0016845553,0.00043444865,0.00055398437,0.0018301951],"category_scores_gemma":[0.0071849865,0.00021454645,0.0010198662,0.0013629762,0.00040328313,0.00084410334,0.0015701349,0.0016122322,0.00043245978],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025686552,0.00073616614,0.89656985,0.0001278199,0.00026192673,0.00016536788,0.0014017601,0.0095946705,0.00061668304,0.0012006098,0.0016307859,0.08743751],"study_design_scores_gemma":[0.0000628135,0.000754127,0.82923096,0.00031364936,0.00029714452,0.00017916152,0.0065963985,0.15037239,0.002179219,0.005666405,0.00425266,0.00009508375],"about_ca_topic_score_codex":0.00618603,"about_ca_topic_score_gemma":0.007529396,"teacher_disagreement_score":0.00618603,"about_ca_system_score_codex":0.00054741144,"about_ca_system_score_gemma":0.001102345,"threshold_uncertainty_score":0.012300074},"labels":[],"label_agreement":null},{"id":"W4413515290","doi":"10.1016/j.stueduc.2025.101512","title":"Stage-based professional development needs of canadian high school teachers: Insights from the dynamic model of educational effectiveness","year":2025,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Teacher Education and Leadership Studies","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université du Québec à Montréal; Université du Québec en Outaouais","funders":"","keywords":"Professional development; Mathematics education; Faculty development; Psychology; Pedagogy","score_opus":0.16209022123376227,"score_gpt":0.4620496655656992,"score_spread":0.29995944433193694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413515290","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9818283,0.00024198309,0.000746459,0.0019846926,0.0000071160157,0.00009812739,0.00014235114,0.000006836133,0.014944152],"genre_scores_gemma":[0.99865675,0.000108543936,0.0003239695,0.000029286899,0.0000010706779,0.00002418508,0.00004183764,0.0000022259069,0.00081201666],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9959766,0.0008595632,0.00013508294,0.00021268641,0.0018333526,0.0009826677],"domain_scores_gemma":[0.9871679,0.0052309968,0.0011703961,0.00031563285,0.0042731953,0.0018418614],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049096625,0.00022876915,0.00041805298,0.0021404894,0.0041985,0.0036426296,0.0019616596,0.0009306303,0.0020014618],"category_scores_gemma":[0.024483742,0.00031944714,0.00034918985,0.002386384,0.0027722085,0.0019210688,0.0016238836,0.0008871336,0.00010199127],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004208517,0.00065901,0.55252117,0.00027277783,0.000058327973,0.00048755092,0.26317593,0.003677647,0.00156829,0.056878872,0.003700448,0.11657917],"study_design_scores_gemma":[0.00002907124,0.00017407082,0.7946155,0.0001327831,0.000044439075,0.00014079058,0.18470147,0.005352943,0.00040985312,0.007864252,0.0064741345,0.000060721206],"about_ca_topic_score_codex":0.9350998,"about_ca_topic_score_gemma":0.9602492,"teacher_disagreement_score":0.9476845,"about_ca_system_score_codex":0.052315455,"about_ca_system_score_gemma":0.06785889,"threshold_uncertainty_score":0.37957698},"labels":[],"label_agreement":null},{"id":"W7104543262","doi":"10.1016/j.stueduc.2025.101530","title":"Stages of teaching expertise from routine to adaptive: A model for advancing teaching effectiveness","year":2025,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Teacher Education and Leadership Studies","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Conceptualization; Perspective (graphical); Focus group; Conceptual framework; Teaching method; Qualitative research; Multimethodology; Resource (disambiguation); Qualitative property; Professional development","score_opus":0.2695009569417753,"score_gpt":0.5581773688283811,"score_spread":0.2886764118866058,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7104543262","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.79975456,0.0015311582,0.10964238,0.0058778566,0.00007686522,0.002590698,0.00035536234,0.00038959525,0.07978158],"genre_scores_gemma":[0.96212375,0.00034561596,0.0358473,0.00009965812,0.0000070883125,0.00040474246,0.00008091175,0.0000145483245,0.0010763892],"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9910252,0.005222564,0.00050222466,0.0006254356,0.0017232309,0.0009013834],"domain_scores_gemma":[0.9841068,0.008702374,0.0017859756,0.0007774504,0.0033297136,0.0012976779],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012274691,0.0007132697,0.00041099623,0.00462292,0.0022679295,0.004616901,0.0016664427,0.0011112266,0.0018659213],"category_scores_gemma":[0.023646008,0.0004959232,0.0008391036,0.0020509523,0.005866442,0.0060943784,0.0032437285,0.0016096882,0.0003002354],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006284483,0.0015796582,0.36730453,0.0017351279,0.00012977331,0.00058437104,0.14301513,0.00994901,0.003880269,0.1923387,0.0024347412,0.27642027],"study_design_scores_gemma":[0.00038710833,0.0068139257,0.5078673,0.00214938,0.000530772,0.0013915149,0.15180372,0.08813436,0.01146299,0.18633726,0.04279386,0.00032781035],"about_ca_topic_score_codex":0.03144831,"about_ca_topic_score_gemma":0.033143576,"teacher_disagreement_score":0.03144831,"about_ca_system_score_codex":0.011389423,"about_ca_system_score_gemma":0.012948642,"threshold_uncertainty_score":0.082636416},"labels":[],"label_agreement":null},{"id":"W7117577854","doi":"10.1016/j.stueduc.2025.101557","title":"Assessing school readiness domains in a large cohort of refugee children: Validation and links with family factors","year":2025,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Early Childhood Education and Development","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Child, Adolescent and Family Mental Health","funders":"Universiti Sains Malaysia; British Academy","keywords":"Rasch model; Refugee; Polytomous Rasch model; Construct validity; Psychometrics; Sample (material); Cohort; Item analysis; Item response theory","score_opus":0.053041866382195935,"score_gpt":0.4417620280352073,"score_spread":0.3887201616530114,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117577854","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9995659,0.000052572188,0.00008055155,0.000018062576,0.0000021902363,0.00003140717,0.000118035234,0.0000012493284,0.00012998049],"genre_scores_gemma":[0.99849856,0.00019217012,0.0005738037,0.000024484472,0.000001772033,0.00013285768,0.00039881413,0.000003493464,0.0001739767],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99924266,0.0002885619,0.00009428501,0.0001053554,0.00013285693,0.00013626904],"domain_scores_gemma":[0.9987212,0.00033487612,0.0003446805,0.00018702104,0.0002628122,0.00014944984],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028332192,0.0005138634,0.00042435923,0.0010330447,0.001035725,0.00081735663,0.00050199253,0.00044272083,0.0010399885],"category_scores_gemma":[0.004597633,0.00034159794,0.00065867545,0.00080648274,0.00056279765,0.00065794075,0.0012863495,0.000932421,0.00032055582],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037986127,0.00007734609,0.99348104,0.00002321605,0.000021057707,0.00015878897,0.0021243996,0.000032510692,0.0003448664,0.000020489635,0.000076881755,0.0036014656],"study_design_scores_gemma":[0.000004095092,0.000119975426,0.9945005,0.000031231328,0.0000098385135,0.00028878404,0.0044832677,0.00007295973,0.00014235618,0.000016515787,0.00032460594,0.000005873965],"about_ca_topic_score_codex":0.0129700815,"about_ca_topic_score_gemma":0.024289848,"teacher_disagreement_score":0.0129700815,"about_ca_system_score_codex":0.0005653026,"about_ca_system_score_gemma":0.001199352,"threshold_uncertainty_score":0.025789201},"labels":[],"label_agreement":null}]}