{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":860,"total_is_capped":false,"direct_labels_cover":1,"predictions_cover":860,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"028605329201","filters":{"topic":"Student Assessment and Feedback"}},"results":[{"id":"W2624412722","doi":"10.3102/0034654316689306","title":"Rethinking the Use of Tests: A Meta-Analysis of Practice Testing","year":2017,"lang":"en","type":"article","venue":"Review of Educational Research","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":536,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Psychology; Meta-analysis; Test (biology); Best practice; Presentation (obstetrics); Mathematics education; Applied psychology; Medicine","authors":[{"name":"Olusola Adesope","is_ca":false},{"name":"Dominic A. Trevisan","is_ca":true},{"name":"NarayanKripa Sundararajan","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.7954275058930537,"gpt":0.643213236651155,"spread":0.1522142692418986,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1569944,0.003208248,0.01433923,0.01233193,0.0008506694,0.006482226,0.004307874,0.002182736,0.001710672],"category_scores_gemma":[0.3911814,0.001964922,0.03975581,0.01126354,0.00230853,0.005236434,0.003169201,0.002894884,0.0002140418],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004970467,"about_ca_system_score_gemma":0.005785631,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006157157,"about_ca_topic_score_gemma":0.01035251,"domain_scores_codex":[0.8246144,0.1119284,0.03956908,0.008198451,0.01480504,0.0008846289],"domain_scores_gemma":[0.6065734,0.3342241,0.02556903,0.0187192,0.01404329,0.0008709473],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"meta_analysis","study_design_gemma":"meta_analysis","study_design_scores_codex":[0.001380233,0.00004416532,0.008096296,0.2169404,0.7289261,0.0001290453,0.0009382486,0.0006800748,0.0003201911,0.0008016152,0.0008352158,0.04090842],"study_design_scores_gemma":[0.0007520349,0.000667019,0.009264308,0.06802141,0.9107859,0.000172118,0.0004297189,0.0005933147,0.0007853201,0.002001567,0.006430806,0.000096601],"study_design_candidate":"meta_analysis","study_design_consensus":"meta_analysis","genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.01732506,0.9678293,0.009367925,0.001242771,0.0008801899,0.001461072,0.0008934905,0.0001187081,0.0008814466],"genre_scores_gemma":[0.5769219,0.3788173,0.03214547,0.002398479,0.0006064837,0.006424193,0.001829189,0.0003264086,0.0005305325],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8430056,"threshold_uncertainty_score":0.8302758,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2885480655","doi":"10.1111/medu.13645","title":"Assessment, feedback and the alchemy of learning","year":2018,"lang":"en","type":"article","venue":"Medical Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":416,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto; Western University","funders":"","keywords":"Formative assessment; Summative assessment; Competence (human resources); CLARITY; Assessment for learning; Judgement; Psychology; Peer feedback; Pedagogy; Medical education; Social psychology; Medicine; Political science","authors":[{"name":"Christopher Watling","is_ca":true},{"name":"Shiphra Ginsburg","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01469089774875304,"gpt":0.3976035401559191,"spread":0.3829126424071661,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03427989,0.0009177558,0.0007305737,0.004510122,0.003587799,0.01433207,0.002501011,0.002963423,0.004550879],"category_scores_gemma":[0.09787574,0.0005379777,0.00072699,0.002459213,0.04284441,0.01535621,0.01194759,0.004247147,0.0007554495],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006716474,"about_ca_system_score_gemma":0.01017538,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004433293,"about_ca_topic_score_gemma":0.003429177,"domain_scores_codex":[0.9347717,0.04565752,0.002721477,0.004146283,0.01157944,0.001123618],"domain_scores_gemma":[0.9077951,0.06714536,0.007652782,0.006909473,0.008267878,0.002229347],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002208004,0.0001167974,0.00637625,0.001967076,0.00008133025,0.0003237024,0.07794037,0.002589702,0.001439867,0.6339277,0.003304239,0.2717122],"study_design_scores_gemma":[0.0001044829,0.0002927089,0.005693409,0.003173785,0.00008953539,0.0005865318,0.01521806,0.002766663,0.002172203,0.8568039,0.1129737,0.0001250445],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1363052,0.03188456,0.3784135,0.1114396,0.001927591,0.0007962334,0.0003116753,0.001021577,0.3379001],"genre_scores_gemma":[0.9262252,0.00444833,0.05946134,0.002688809,0.0003038728,0.0004908401,0.00005996537,0.0001441219,0.006177449],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03427989,"threshold_uncertainty_score":0.1812915,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2183435816","doi":"10.1007/s11092-015-9233-6","title":"Teacher assessment literacy: a review of international standards and measures","year":2015,"lang":"en","type":"review","venue":"Educational Assessment Evaluation and Accountability","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":253,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Literacy; Educational assessment; Mainland; Psychology; Pedagogy; Geography","authors":[{"name":"Christopher DeLuca","is_ca":true},{"name":"Danielle LaPointe-McEwan","is_ca":true},{"name":"Ulemu Luhanga","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1374248929728662,"gpt":0.5795117938372106,"spread":0.4420869008643444,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02518895,0.001299449,0.005311014,0.01795497,0.0008460883,0.003551908,0.003326966,0.002146365,0.002502413],"category_scores_gemma":[0.06414508,0.0008883267,0.002540295,0.01568116,0.00238161,0.004147775,0.002843651,0.002425466,0.0006287313],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004745263,"about_ca_system_score_gemma":0.02158045,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01099142,"about_ca_topic_score_gemma":0.01972708,"domain_scores_codex":[0.9841541,0.003658684,0.006487186,0.001101896,0.004334907,0.0002631819],"domain_scores_gemma":[0.9449297,0.03647909,0.007558456,0.0009664416,0.009522762,0.0005435169],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0001248835,0.00007546182,0.002158733,0.1536993,0.0005238173,0.00005114418,0.0004648533,0.0002419511,0.0002542209,0.004434596,0.01328881,0.8246823],"study_design_scores_gemma":[0.0001099025,0.0002391511,0.01481729,0.4567488,0.005636623,0.0007198242,0.00108675,0.0003301087,0.001077679,0.005299672,0.5137883,0.0001459177],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0002492683,0.9972396,0.0005561354,0.0008709863,0.000168665,0.00007580891,0.0002136282,0.00001154282,0.0006142973],"genre_scores_gemma":[0.003069689,0.9930817,0.002676064,0.0005098516,0.0001171536,0.0001618511,0.0002523741,0.000008908911,0.0001223857],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.02518895,"threshold_uncertainty_score":0.1332135,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2167980328","doi":"10.1007/s10459-010-9263-2","title":"Exploring the divergence between self-assessment and self-monitoring","year":2010,"lang":"en","type":"article","venue":"Advances in Health Sciences Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":190,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia; University of British Columbia Hospital","funders":"","keywords":"Self-assessment; Self-monitoring; Self evaluation; Psychology; Medicine; Applied psychology; Social psychology","authors":[{"name":"Kevin W. Eva","is_ca":true},{"name":"Glenn Regehr","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06736112885207077,"gpt":0.4562451782461324,"spread":0.3888840493940616,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04706091,0.0003800197,0.0006694117,0.00282714,0.0005658582,0.0052421,0.001086097,0.001009539,0.000855959],"category_scores_gemma":[0.1964387,0.0003151138,0.0004425517,0.001709631,0.004966322,0.004706134,0.003886856,0.002053051,0.000132605],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001169547,"about_ca_system_score_gemma":0.001411198,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001688461,"about_ca_topic_score_gemma":0.001423639,"domain_scores_codex":[0.9602835,0.02880119,0.001760899,0.002302935,0.00623793,0.0006136644],"domain_scores_gemma":[0.7310295,0.2281751,0.01510182,0.01022615,0.01352436,0.001943087],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"qualitative","study_design_scores_codex":[0.0004220678,0.0003916098,0.5219948,0.0007004667,0.000362208,0.0002317726,0.1216485,0.002495443,0.002625848,0.08830694,0.0003922441,0.2604281],"study_design_scores_gemma":[0.00004481295,0.001098424,0.7935727,0.001028497,0.000139275,0.0009835673,0.0473479,0.02242898,0.00372509,0.1207607,0.008621898,0.0002482274],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8895302,0.002012019,0.07161687,0.001882581,0.0000681575,0.0001546987,0.00006970538,0.00006811202,0.0345976],"genre_scores_gemma":[0.9944149,0.0001734755,0.004946624,0.0001057729,0.00001217398,0.0000543223,0.00002113185,0.0000104553,0.0002610098],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04706091,"threshold_uncertainty_score":0.2488849,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2159609851","doi":"10.1191/0265532204lt288oa","title":"ESL/EFL instructors’ classroom assessment practices: purposes, methods, and procedures","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":184,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Alberta; Queen's University","funders":"","keywords":"Psychology; English as a foreign language; Mathematics education; Tertiary level; Pedagogy; Second language; Language assessment; Linguistics","authors":[{"name":"Liying Cheng","is_ca":true},{"name":"Todd Rogers","is_ca":true},{"name":"Huiqin Hu","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06401151606627827,"gpt":0.4566933338742858,"spread":0.3926818178080075,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04419564,0.0005909574,0.0005557169,0.004196423,0.00248741,0.002311585,0.001412967,0.0005317276,0.002014607],"category_scores_gemma":[0.07329158,0.0004183087,0.0002716429,0.002983906,0.002312854,0.001303968,0.003125248,0.0007570963,0.001132535],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004144968,"about_ca_system_score_gemma":0.006230204,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01155767,"about_ca_topic_score_gemma":0.02301954,"domain_scores_codex":[0.9547526,0.02701685,0.005813431,0.002833513,0.007963849,0.001619732],"domain_scores_gemma":[0.9352966,0.02213053,0.006743305,0.006459848,0.02684863,0.00252109],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.000888226,0.002255439,0.2514513,0.0007885541,0.00002985677,0.0003699488,0.1596395,0.0004672911,0.0112482,0.001081218,0.00196139,0.5698192],"study_design_scores_gemma":[0.0003243159,0.002467649,0.7964455,0.001025194,0.00006302998,0.0005444374,0.1236496,0.00307432,0.03002802,0.001700874,0.04046831,0.0002087711],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9607155,0.0003306784,0.01445856,0.0004667196,0.00003704431,0.007025859,0.0004268307,0.0002383994,0.01630036],"genre_scores_gemma":[0.9315376,0.0003620019,0.052952,0.0002430971,0.00003468316,0.01053553,0.0002586859,0.00005263382,0.00402367],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04419564,"threshold_uncertainty_score":0.2337317,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2108604890","doi":"10.1177/1478210314566733","title":"International trends in the implementation of assessment for learning: Implications for policy and practice","year":2015,"lang":"en","type":"article","venue":"Policy Futures in Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":179,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"Brock University; Queen's University","funders":"","keywords":"Summative assessment; Globe; Policy learning; Political science; Educational assessment; International comparisons; Public administration; Policy analysis; Economic growth; Sociology; Regional science; Formative assessment; Pedagogy; Economics; Psychology","authors":[{"name":"Menucha Birenbaum","is_ca":false},{"name":"Christopher DeLuca","is_ca":true},{"name":"Lorna Earl","is_ca":false},{"name":"Margaret Heritage","is_ca":false},{"name":"Val Klenowski","is_ca":false},{"name":"Anne Looney","is_ca":false},{"name":"Kari Smith","is_ca":false},{"name":"Helen Timperley","is_ca":false},{"name":"Louis Volante","is_ca":true},{"name":"Claire Wyatt‐Smith","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0812123857192674,"gpt":0.5733258516846382,"spread":0.4921134659653709,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05351208,0.0003104572,0.0005905159,0.003489332,0.002340505,0.01247362,0.002220695,0.004459252,0.007039082],"category_scores_gemma":[0.0930329,0.0003343809,0.0005830289,0.00878365,0.008317293,0.0136933,0.006555142,0.008501506,0.0008197036],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01808951,"about_ca_system_score_gemma":0.03239688,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03544124,"about_ca_topic_score_gemma":0.02222487,"domain_scores_codex":[0.974506,0.0109554,0.003243674,0.002786248,0.005466879,0.003041816],"domain_scores_gemma":[0.848447,0.08177641,0.01767701,0.005455644,0.03640116,0.01024279],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00014797,0.0003864718,0.04241535,0.002450489,0.00003762682,0.0001553286,0.01531364,0.001295037,0.0005368834,0.4134339,0.03244552,0.4913818],"study_design_scores_gemma":[0.00005596675,0.0004262863,0.146819,0.01349752,0.00004022954,0.0005637968,0.07023889,0.001755272,0.001244928,0.1080193,0.6571845,0.0001542326],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.04712626,0.05032902,0.005721544,0.8226156,0.002096516,0.0001102267,0.0004108596,0.0001357884,0.07145417],"genre_scores_gemma":[0.8506019,0.06431985,0.0152095,0.05997874,0.001523332,0.0002653089,0.0005321481,0.0001541199,0.007415115],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.05351208,"threshold_uncertainty_score":0.2830023,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1520074575","doi":"10.19173/irrodl.v15i3.1680","title":"Peer assessment for massive open online courses (MOOCs)","year":2014,"lang":"en","type":"article","venue":"The International Review of Research in Open and Distributed Learning","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":178,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Formative assessment; Summative assessment; Peer assessment; Computer science; Peer feedback; Credibility; Assessment for learning; Online assessment; Peer evaluation; Open education; Credentialing; Multimedia; Higher education; World Wide Web; Medical education; Mathematics education; Psychology","authors":[{"name":"Hoi K. Suen","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1579032256599237,"gpt":0.5635018527575003,"spread":0.4055986270975765,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0297819,0.0006980564,0.001151335,0.004154005,0.001763219,0.004097313,0.002191184,0.001180066,0.007644879],"category_scores_gemma":[0.170088,0.0003504959,0.0005834828,0.002934451,0.001397992,0.00375019,0.00374148,0.001481674,0.003235969],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001515279,"about_ca_system_score_gemma":0.004187301,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003903937,"about_ca_topic_score_gemma":0.003893173,"domain_scores_codex":[0.9479783,0.03003401,0.001695673,0.002015815,0.01769348,0.0005826119],"domain_scores_gemma":[0.8876217,0.06727206,0.006905161,0.007554996,0.02777615,0.002870013],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0002637147,0.0002541546,0.005731065,0.00102361,0.00007795972,0.00007653968,0.001314431,0.002282988,0.001010652,0.01295848,0.009551789,0.9654546],"study_design_scores_gemma":[0.0008199272,0.006573712,0.1365505,0.00733013,0.0006725159,0.001682325,0.0128483,0.1819897,0.02161776,0.2392993,0.3898148,0.0008009856],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1610495,0.03288652,0.6215609,0.007488229,0.004377466,0.00465818,0.0007238938,0.004820632,0.1624347],"genre_scores_gemma":[0.7628887,0.007179899,0.2106116,0.0003530381,0.001010248,0.001725135,0.0005947043,0.0004526019,0.01518388],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9702181,"threshold_uncertainty_score":0.1575036,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2105451781","doi":"10.1191/0265532204lt287oa","title":"Teacher formative assessment and talk in classroom contexts: assessment as discourse and assessment of discourse","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":159,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Formative assessment; Psychology; Pedagogy; Discourse analysis; Applied linguistics; Mathematics education; Assessment for learning; Systemic functional linguistics; Linguistics","authors":[{"name":"Constant Leung","is_ca":false},{"name":"Bernard Mohan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03032002706073061,"gpt":0.4286908790492009,"spread":0.3983708519884703,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03478556,0.000617564,0.0007696394,0.005578764,0.00147686,0.01191711,0.001677766,0.001498957,0.001622202],"category_scores_gemma":[0.1005457,0.0003586864,0.0003531172,0.003686693,0.01388191,0.009962572,0.005300997,0.002310413,0.0002805237],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00229644,"about_ca_system_score_gemma":0.003387991,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001555785,"about_ca_topic_score_gemma":0.002265697,"domain_scores_codex":[0.9522936,0.03846737,0.001698786,0.001208084,0.005936554,0.0003956448],"domain_scores_gemma":[0.9081358,0.07414794,0.007252437,0.00433499,0.004460025,0.001668874],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001970836,0.0004703581,0.05181728,0.001429481,0.00009859348,0.000492739,0.4369698,0.00219146,0.006109028,0.1287884,0.001066504,0.3703692],"study_design_scores_gemma":[0.0001288726,0.001841601,0.1639504,0.003538027,0.0001551987,0.004283031,0.3164106,0.02057525,0.01895963,0.4026691,0.06704799,0.0004403103],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5576613,0.01008833,0.3531671,0.006669908,0.0003292862,0.0008942871,0.0001775124,0.000387094,0.07062522],"genre_scores_gemma":[0.9515393,0.001755937,0.04319308,0.0001705102,0.00008937758,0.0006560503,0.00004003309,0.00005014699,0.002505473],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03478556,"threshold_uncertainty_score":0.1839659,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2112485328","doi":"10.1080/15434303.2015.1010726","title":"Teachers’ Grading Decision Making: Multiple Influencing Factors and Methods","year":2015,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":138,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Grading (engineering); Psychology; Mathematics education; English language; Multivariate analysis of variance; Statistics; Engineering; Mathematics","authors":[{"name":"Liying Cheng","is_ca":true},{"name":"Youyi Sun","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04392888403571169,"gpt":0.4426289617588582,"spread":0.3987000777231465,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01689852,0.0009483484,0.0008301566,0.003332958,0.001701881,0.003961886,0.0009623157,0.0004951389,0.002809022],"category_scores_gemma":[0.04934454,0.0005961421,0.001144686,0.002893308,0.001163524,0.001162322,0.001340243,0.0007221554,0.0002582053],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002369377,"about_ca_system_score_gemma":0.004415166,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01248928,"about_ca_topic_score_gemma":0.01599806,"domain_scores_codex":[0.9792632,0.008851162,0.00219875,0.002247133,0.006202952,0.001236692],"domain_scores_gemma":[0.9174977,0.05241028,0.0152197,0.003204464,0.008862635,0.002805194],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001997072,0.0002141826,0.9457048,0.0001680297,0.0002129289,0.0002285701,0.01221618,0.0004780262,0.0007314763,0.000487487,0.0003178524,0.03904076],"study_design_scores_gemma":[0.00003721968,0.0002177043,0.9736769,0.0001558359,0.0003459095,0.0002730929,0.01167925,0.008557175,0.001382355,0.001270156,0.002310841,0.00009368276],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9835622,0.0004202673,0.009157084,0.0002548019,0.0000371652,0.000571558,0.0001365279,0.00007498518,0.005785475],"genre_scores_gemma":[0.9944124,0.00008485719,0.004606484,0.00002219572,0.00001150846,0.0001374426,0.00006755045,0.00001428714,0.0006432816],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01689852,"threshold_uncertainty_score":0.089369,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1600384676","doi":"","title":"Teaching to the Test: What Every Educator and Policy-Maker Should Know.","year":2004,"lang":"en","type":"article","venue":"Canadian Journal of Educational Administration and Policy","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":125,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Standardized test; Mathematics education; Test (biology); Curriculum; Psychology; Strengths and weaknesses; Norm (philosophy); Medical education; Pedagogy; Medicine; Social psychology; Political science","authors":[{"name":"Louis Volante","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03360182513708301,"gpt":0.3962923616583372,"spread":0.3626905365212542,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03366942,0.001718247,0.00210904,0.004486408,0.003204187,0.00878798,0.005556127,0.01605866,0.01494488],"category_scores_gemma":[0.1273872,0.0007285231,0.001013456,0.003302932,0.009310993,0.02067927,0.004837667,0.0153413,0.0166321],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004943279,"about_ca_system_score_gemma":0.02129801,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01608302,"about_ca_topic_score_gemma":0.01697704,"domain_scores_codex":[0.9794601,0.009183844,0.001924232,0.0008506429,0.007462112,0.001119216],"domain_scores_gemma":[0.8841751,0.03937459,0.005185407,0.008082137,0.04538275,0.01779994],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004374008,0.0001442192,0.002664431,0.0006657792,0.00002151162,0.0002569167,0.0004757501,0.0001267041,0.0001197968,0.006026803,0.6614284,0.3280258],"study_design_scores_gemma":[0.00005450557,0.0001545744,0.004829085,0.01210757,0.00004535106,0.001597635,0.003698803,0.000363933,0.0003125464,0.05026188,0.9264517,0.0001224354],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.0006243557,0.06138041,0.006763033,0.8903543,0.02435288,0.0001701632,0.0005668545,0.0006338034,0.01515427],"genre_scores_gemma":[0.02784802,0.1915692,0.05231432,0.6442623,0.04625358,0.001569793,0.002555343,0.001407009,0.03222048],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03366942,"threshold_uncertainty_score":0.1780631,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2116070328","doi":"10.1177/0265532210376379","title":"Think-aloud protocols in research on essay rating: An empirical study of their veridicality and reactivity","year":2010,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":121,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"","keywords":"Think aloud protocol; Psychology; Protocol analysis; Perception; Empirical research; Sample (material); Qualitative research; Rating scale; Social psychology; Nomothetic and idiographic; Cognitive psychology; Applied psychology; Developmental psychology; Epistemology; Cognitive science","authors":[{"name":"Khaled Barkaoui","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2923764194054679,"gpt":0.5510859877479262,"spread":0.2587095683424582,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03743877,0.0007361121,0.0004927889,0.00120755,0.0009173541,0.001642874,0.001102275,0.0009466977,0.00116799],"category_scores_gemma":[0.2199901,0.000555056,0.000320532,0.001070857,0.001426186,0.001562504,0.001957292,0.001280929,0.0006609897],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004902453,"about_ca_system_score_gemma":0.0006471811,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001933144,"about_ca_topic_score_gemma":0.0002665671,"domain_scores_codex":[0.9395154,0.04742792,0.003363881,0.002915364,0.006367211,0.0004101973],"domain_scores_gemma":[0.6944738,0.2487216,0.0227628,0.01710609,0.01570473,0.001230987],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00291203,0.001885879,0.1145801,0.002211474,0.0003552911,0.0007919685,0.2230306,0.001853537,0.1190248,0.005478701,0.001786692,0.526089],"study_design_scores_gemma":[0.0007464606,0.02040762,0.5165713,0.002543318,0.0005755922,0.007781459,0.1235017,0.02727211,0.2092096,0.0289061,0.06154568,0.0009391778],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8939719,0.000418444,0.09813229,0.0001922338,0.0001273409,0.001318618,0.000135502,0.0002361061,0.005467661],"genre_scores_gemma":[0.913073,0.0005242794,0.07989421,0.0002689874,0.0001163691,0.003634382,0.0001949199,0.0001571192,0.002136816],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9625612,"threshold_uncertainty_score":0.1979975,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1986967177","doi":"10.3109/0142159x.2010.486063","title":"Assessment steers learning down the right road: Impact of progress testing on licensing examination performance","year":2010,"lang":"en","type":"article","venue":"Medical Teacher","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":121,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto; McMaster University","funders":"","keywords":"Formative assessment; Test (biology); Curriculum; Psychology; Medicine; Medical education; Mathematics education; Pedagogy","authors":[{"name":"Geoff Norman","is_ca":true},{"name":"Alan J. Neville","is_ca":true},{"name":"Jennifer Blake","is_ca":true},{"name":"Barber Mueller","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02780155950402934,"gpt":0.3811736229013911,"spread":0.3533720633973618,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007991611,0.0004973412,0.0005831275,0.0005487091,0.000367403,0.0008742223,0.0005972525,0.0006704116,0.002911099],"category_scores_gemma":[0.07465793,0.0001874119,0.0005346778,0.0003550385,0.0005483759,0.0008385214,0.001174903,0.001257822,0.0005460425],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005115114,"about_ca_system_score_gemma":0.00132378,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002248702,"about_ca_topic_score_gemma":0.002345902,"domain_scores_codex":[0.9869612,0.008533501,0.0004337691,0.0004776058,0.003132275,0.0004615418],"domain_scores_gemma":[0.9398135,0.04391813,0.006395129,0.001958564,0.002591866,0.005322757],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01251943,0.01483534,0.1348195,0.00043973,0.0002945124,0.000347228,0.001473943,0.004221201,0.007969149,0.0003753449,0.002780956,0.8199238],"study_design_scores_gemma":[0.0007951513,0.07349762,0.9048786,0.0002487183,0.0002998528,0.000510987,0.0009529607,0.005608432,0.009147123,0.000582253,0.003382654,0.00009567889],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9915259,0.0007314097,0.0009144827,0.0009649775,0.0001038219,0.00009139469,0.00004461602,0.0001641151,0.005459228],"genre_scores_gemma":[0.99777,0.0001937752,0.0008546276,0.00006814484,0.0000358567,0.00002070917,0.00003115403,0.00001228209,0.001013618],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007991611,"threshold_uncertainty_score":0.04226416,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2136543334","doi":"10.3138/cmlr.64.1.135","title":"A New Washback Model of Students’ Learning","year":2007,"lang":"en","type":"article","venue":"Canadian Modern Language Review/ La Revue canadienne des langues vivantes","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":119,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Test (biology); Mathematics education; Context (archaeology); Psychology; English as a foreign language; Pedagogy; Geography","authors":[{"name":"Chih‐Min Shih","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02422845520616734,"gpt":0.3123421666566812,"spread":0.2881137114505139,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007589388,0.0009604865,0.0007037085,0.003069831,0.001101455,0.003809516,0.003144868,0.001944372,0.008910361],"category_scores_gemma":[0.03173077,0.0006420445,0.00092838,0.001255325,0.003807114,0.007226914,0.003237752,0.001851499,0.001282171],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004749323,"about_ca_system_score_gemma":0.003105505,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005673409,"about_ca_topic_score_gemma":0.002535722,"domain_scores_codex":[0.9917897,0.003318833,0.0003507255,0.001118678,0.0029103,0.0005119099],"domain_scores_gemma":[0.9828898,0.00816076,0.001877433,0.002708352,0.003613665,0.0007500549],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001557138,0.004058977,0.2287485,0.0008378538,0.0002422042,0.001604907,0.08747581,0.01602173,0.01652091,0.1696776,0.004226625,0.4690278],"study_design_scores_gemma":[0.0004157505,0.007213834,0.2819866,0.0007746168,0.0003396618,0.00235307,0.06681259,0.3363887,0.01929439,0.257735,0.026321,0.0003647832],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7858148,0.0004787008,0.1290372,0.003083152,0.0001207054,0.000937501,0.0002548685,0.001245707,0.07902728],"genre_scores_gemma":[0.9835218,0.00005471726,0.01281274,0.0001038305,0.000008790301,0.0002342928,0.00005423354,0.00004296549,0.003166637],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008910361,"threshold_uncertainty_score":0.04013699,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3191701502","doi":"10.1080/0142159x.2021.1957088","title":"Ottawa 2020 consensus statement for programmatic assessment – 1. Agreement on the principles","year":2021,"lang":"en","type":"article","venue":"Medical Teacher","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":115,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"The Wilson Centre; University of Toronto","funders":"","keywords":"Operationalization; Medical education; Curriculum; Program evaluation; Engineering ethics; Medicine; Psychology; Political science; Pedagogy; Engineering","authors":[{"name":"Sylvia Heeneman","is_ca":false},{"name":"Lubberta H. de Jong","is_ca":false},{"name":"Luke Dawson","is_ca":false},{"name":"Tim Wilkinson","is_ca":false},{"name":"Anna Ryan","is_ca":false},{"name":"Glendon R. Tait","is_ca":true},{"name":"Neil Rice","is_ca":false},{"name":"Dario Torre","is_ca":false},{"name":"Adrian Freeman","is_ca":false},{"name":"Cees van der Vleuten","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07826089378203037,"gpt":0.4097405264132591,"spread":0.3314796326312287,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2144215,0.00212013,0.004524002,0.01080384,0.007263087,0.01426928,0.01862385,0.01926865,0.01260156],"category_scores_gemma":[0.3071682,0.004027469,0.009620917,0.008445152,0.01199296,0.004734938,0.01322877,0.02617141,0.01189552],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.04919924,"about_ca_system_score_gemma":0.2153567,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.193291,"about_ca_topic_score_gemma":0.1557359,"domain_scores_codex":[0.7150252,0.1374164,0.06237194,0.00745107,0.0657317,0.01200367],"domain_scores_gemma":[0.5684406,0.102334,0.02022155,0.01761128,0.2719711,0.01942149],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003582635,0.0001548976,0.001668052,0.01107148,0.0002830783,0.0003435674,0.003306563,0.001520199,0.0004775359,0.0472433,0.7994764,0.1340967],"study_design_scores_gemma":[0.0001703602,0.00008450548,0.003794283,0.02919088,0.0002283945,0.000271323,0.00162152,0.0006241713,0.0006254831,0.01904741,0.9441017,0.0002399061],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"other","genre_scores_codex":[0.003037771,0.06233248,0.1568066,0.5145002,0.0572215,0.02979904,0.01889674,0.002378592,0.1550272],"genre_scores_gemma":[0.06376386,0.04369012,0.5508122,0.1029831,0.005797138,0.07985347,0.02922631,0.002400986,0.1214727],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.7855785,"threshold_uncertainty_score":0.9687582,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2148189517","doi":"10.1111/j.1365-2729.2008.00290.x","title":"Peering into large lectures: examining peer and expert mark agreement using peerScholar, an online peer assessment tool","year":2008,"lang":"en","type":"article","venue":"Journal of Computer Assisted Learning","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":114,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"The Scarborough Hospital; University of Toronto","funders":"","keywords":"Peer assessment; Peering; Class (philosophy); Computer science; Accountability; Class size; Peer evaluation; Rank (graph theory); Peer feedback; Mathematics education; Online assessment; World Wide Web; Multimedia; Higher education; The Internet; Psychology; Artificial intelligence; Formative assessment; Mathematics","authors":[{"name":"Dwayne E. Paré","is_ca":true},{"name":"Steve Joordens","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08174915342893448,"gpt":0.3820792831741884,"spread":0.3003301297452539,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01938182,0.0003390672,0.000660887,0.001891392,0.001266709,0.001481121,0.0009699783,0.0005243614,0.002219732],"category_scores_gemma":[0.09200644,0.0003050565,0.0003083428,0.0005282774,0.0009131139,0.001509397,0.002154832,0.0007388576,0.0006977096],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005549751,"about_ca_system_score_gemma":0.0006777181,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009312224,"about_ca_topic_score_gemma":0.001650244,"domain_scores_codex":[0.9814222,0.01109448,0.001358363,0.001494109,0.004117919,0.0005129417],"domain_scores_gemma":[0.9035031,0.05862084,0.0118787,0.005536868,0.01702445,0.00343613],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00347568,0.006326817,0.5271579,0.0005042432,0.0002908446,0.0005329011,0.1171263,0.001374296,0.02612616,0.001198346,0.002769066,0.3131174],"study_design_scores_gemma":[0.0003606793,0.01437834,0.8344499,0.0002219126,0.0002072077,0.0009646505,0.08579387,0.01684983,0.03647069,0.002337336,0.00764474,0.0003207423],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9969146,0.00002404485,0.001758699,0.00002798454,0.00001204617,0.0001418647,0.00001882563,0.00002845121,0.001073477],"genre_scores_gemma":[0.9950936,0.0000370403,0.003477926,0.00002953088,0.00001316328,0.0002469599,0.0000392203,0.00001279239,0.00104981],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9806182,"threshold_uncertainty_score":0.1025021,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2523280788","doi":"10.1080/10627197.2016.1236677","title":"Approaches to Classroom Assessment Inventory: A New Instrument to Support Teacher Assessment Literacy","year":2016,"lang":"en","type":"article","venue":"Educational Assessment","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":112,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Accountability; Literacy; Construct (python library); Educational assessment; Psychology; Construct validity; Mathematics education; Standards for Educational and Psychological Testing; Standardized test; Standards-based assessment; Process (computing); Authentic assessment; Pedagogy; Psychometrics; Higher education; Computer science; Political science; Education theory; Curriculum","authors":[{"name":"Christopher DeLuca","is_ca":true},{"name":"Danielle LaPointe-McEwan","is_ca":true},{"name":"Ulemu Luhanga","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1328510477980371,"gpt":0.4138136676963231,"spread":0.280962619898286,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008253588,0.000430441,0.0005822259,0.003858602,0.0009327062,0.001996543,0.001059876,0.0004075276,0.003060797],"category_scores_gemma":[0.03430567,0.000440701,0.0006528134,0.002093418,0.0008376984,0.003129363,0.003665721,0.002538804,0.001466707],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001490116,"about_ca_system_score_gemma":0.005966381,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003421281,"about_ca_topic_score_gemma":0.009186864,"domain_scores_codex":[0.9914857,0.003093509,0.001549962,0.000520675,0.00306019,0.000290035],"domain_scores_gemma":[0.9712158,0.01304952,0.004314045,0.002022656,0.007819675,0.001578296],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001697582,0.001226021,0.1660303,0.0004491439,0.00008644386,0.0001909056,0.01123625,0.0008948211,0.004672242,0.006215385,0.02711555,0.7817132],"study_design_scores_gemma":[0.0002605428,0.001775181,0.6868221,0.0009772708,0.0002240202,0.001873028,0.01090519,0.01124982,0.006550833,0.01668364,0.2623265,0.0003519336],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5186203,0.003280367,0.3064511,0.008135222,0.001254906,0.01422834,0.01033379,0.009342569,0.1283535],"genre_scores_gemma":[0.3423016,0.001643281,0.625768,0.0009850942,0.0002526271,0.01355159,0.003683569,0.0005600009,0.01125427],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008253588,"threshold_uncertainty_score":0.04364967,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2138588257","doi":"10.18806/tesl.v27i2.1052","title":"A Peer Review Training Workshop: Coaching Students to Give and Evaluate Peer Feedback","year":2010,"lang":"en","type":"review","venue":"TESL Canada Journal","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":109,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Peer feedback; Coaching; Psychology; Peer review; Training (meteorology); Consciousness; Pedagogy; Medical education; Mathematics education; Political science","authors":[{"name":"Ricky Lam","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1447572715000492,"gpt":0.4614545996722544,"spread":0.3166973281722052,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05273046,0.001061499,0.001611779,0.002808115,0.002156302,0.001637124,0.003505669,0.002443965,0.003564148],"category_scores_gemma":[0.0845205,0.000621153,0.000921305,0.001190679,0.0014435,0.001642455,0.002760922,0.002532561,0.003244256],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008463947,"about_ca_system_score_gemma":0.006546143,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001135086,"about_ca_topic_score_gemma":0.004050822,"domain_scores_codex":[0.9506447,0.03364304,0.00328274,0.001515012,0.0102504,0.0006640361],"domain_scores_gemma":[0.9137431,0.04544185,0.008969618,0.004948833,0.02188016,0.005016357],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004227665,0.0009900694,0.001368068,0.005858953,0.0001928425,0.000589496,0.004970878,0.0003523078,0.007774157,0.00110343,0.05736652,0.9190105],"study_design_scores_gemma":[0.001740231,0.007786678,0.02698425,0.01229248,0.0005474478,0.007515544,0.007177745,0.002843565,0.01632604,0.00559302,0.9107158,0.0004772104],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.1196646,0.1719428,0.4207239,0.07823939,0.03209262,0.06855734,0.0005211683,0.006673516,0.1015847],"genre_scores_gemma":[0.2617332,0.1186377,0.5475936,0.009429468,0.00788373,0.03229143,0.0004873861,0.000345561,0.02159802],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.9472696,"threshold_uncertainty_score":0.2788686,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2897106912","doi":"10.1111/medu.13746","title":"When I say … feedback","year":2018,"lang":"en","type":"article","venue":"Medical Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":108,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Constructive; Process (computing); Term (time); Space (punctuation); Series (stratigraphy); Psychology; Computer science; Social psychology; Sociology; Physics; Programming language","authors":[{"name":"Rola Ajjawi","is_ca":false},{"name":"Glenn Regehr","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01926714578388553,"gpt":0.3851389734389596,"spread":0.365871827655074,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01105968,0.001226141,0.0008041614,0.001296925,0.003079024,0.005668757,0.001185501,0.003545238,0.05547263],"category_scores_gemma":[0.08249448,0.0004670931,0.0008192706,0.0006645034,0.003430336,0.006549837,0.004765214,0.009461005,0.05299287],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001674452,"about_ca_system_score_gemma":0.003777085,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002022638,"about_ca_topic_score_gemma":0.004630316,"domain_scores_codex":[0.9841573,0.006582563,0.0008658708,0.0008089949,0.006470648,0.001114668],"domain_scores_gemma":[0.963422,0.01210018,0.002160475,0.001593194,0.01432575,0.006398456],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"qualitative","study_design_scores_codex":[0.00004358937,0.00003079469,0.0002081265,0.0001190229,0.000005812642,0.00003200562,0.001286663,0.0000226502,0.0002869594,0.002308949,0.9363122,0.05934323],"study_design_scores_gemma":[0.00001182997,0.00005688006,0.0005430353,0.0002250676,0.000006544767,0.00009611047,0.00151861,0.00006601337,0.0004182155,0.00226038,0.994764,0.0000333599],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.002979546,0.006835451,0.02137217,0.4567207,0.2492731,0.0003086002,0.0005851209,0.006807533,0.2551178],"genre_scores_gemma":[0.07098896,0.01144715,0.01989164,0.3055582,0.05821643,0.0008113228,0.0008609443,0.003684855,0.5285404],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.05547263,"threshold_uncertainty_score":0.1855744,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2122826394","doi":"10.5539/ass.v4n3p78","title":"Student Perceptions and Preferences for Feedback","year":2009,"lang":"en","type":"article","venue":"Asian Social Science","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":107,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"Macquarie University","keywords":"ROWE; Quality (philosophy); Perception; Psychology; Key (lock); Undergraduate research; Element (criminal law); Medical education; Mathematics education; Computer science; Marketing; Political science; Business; Medicine","authors":[{"name":"Anna Rowe","is_ca":false},{"name":"Leigh Wood","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0368983349792774,"gpt":0.392057273630894,"spread":0.3551589386516166,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006881215,0.0002825104,0.000461199,0.001116496,0.0007879096,0.002406073,0.0003979595,0.0009437928,0.003390625],"category_scores_gemma":[0.04936105,0.0001727008,0.0005955115,0.0006476321,0.0006184676,0.0007241374,0.001030669,0.001059635,0.0005469775],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007197597,"about_ca_system_score_gemma":0.0008177535,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001615206,"about_ca_topic_score_gemma":0.001460499,"domain_scores_codex":[0.9919969,0.003372871,0.0006739202,0.0003704451,0.00271547,0.0008703204],"domain_scores_gemma":[0.9602551,0.02091036,0.00562533,0.0008320428,0.007279055,0.005098179],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"qualitative","study_design_scores_codex":[0.001496792,0.00147422,0.7512479,0.0004242794,0.0001807339,0.0009681128,0.09319429,0.0005883319,0.008681918,0.000548968,0.002471594,0.1387229],"study_design_scores_gemma":[0.0001290006,0.005507839,0.7857044,0.0002846317,0.0001211774,0.002461288,0.1858039,0.002059967,0.003970334,0.0008473928,0.01292187,0.0001882841],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9968521,0.0001178625,0.0003777481,0.0004045533,0.00002081389,0.00002569263,0.00003889555,0.00001434775,0.002148049],"genre_scores_gemma":[0.9987238,0.00009610568,0.00025435,0.0001137592,0.00001260356,0.00001908688,0.00003162345,0.00000501289,0.0007435717],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006881215,"threshold_uncertainty_score":0.03639179,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2479837187","doi":"10.1007/978-3-319-39211-0","title":"Assessment for Learning: Meeting the Challenge of Implementation","year":2016,"lang":"en","type":"book","venue":"The enabling power of assessment","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":105,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Engineering ethics; Computer science; Engineering management; Engineering","authors":[{"name":"Linda Allal","is_ca":true},{"name":"Dany Laveault","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04101925382497879,"gpt":0.3993390365835251,"spread":0.3583197827585463,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0541296,0.0009734,0.001514851,0.001873669,0.002828834,0.02083368,0.004421146,0.008056941,0.009620931],"category_scores_gemma":[0.09115326,0.000617428,0.0007151967,0.001582838,0.01759369,0.02514215,0.01106318,0.01656519,0.003911722],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006355197,"about_ca_system_score_gemma":0.02890906,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00484491,"about_ca_topic_score_gemma":0.005924587,"domain_scores_codex":[0.9618586,0.02258993,0.001614517,0.001498802,0.01148748,0.0009506674],"domain_scores_gemma":[0.8858657,0.09141654,0.001743058,0.005697603,0.01128707,0.00398996],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001947694,0.00008947172,0.0003433909,0.00070652,0.00001562875,0.00004254864,0.002173063,0.0006409662,0.0002247203,0.4577105,0.09436889,0.4436648],"study_design_scores_gemma":[0.00002420359,0.00009272576,0.0005747371,0.00241072,0.00001042658,0.000128754,0.002245408,0.001425257,0.0003165352,0.5762968,0.4164321,0.00004228651],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"other","genre_scores_codex":[0.002890042,0.0617509,0.1456687,0.5755782,0.0102428,0.0003437689,0.0001137112,0.001248915,0.202163],"genre_scores_gemma":[0.2527533,0.09619332,0.416793,0.07290521,0.0157136,0.002274571,0.0003101121,0.001514296,0.1415426],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.0541296,"threshold_uncertainty_score":0.2862681,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2568968076","doi":"10.1177/1362168816684366","title":"Developing the assessment literacy of teachers in Chinese language classrooms: A focus on assessment task design","year":2017,"lang":"en","type":"article","venue":"Language Teaching Research","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":103,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto; University of Calgary","funders":"","keywords":"Task (project management); Literacy; Psychology; Mathematics education; Professional development; Pedagogy; Task analysis; Authentic assessment; Quality (philosophy); Faculty development; Language assessment; Curriculum","authors":[{"name":"Kim Koh","is_ca":true},{"name":"Lydia E Carol-Ann Burke","is_ca":true},{"name":"Allan Luke","is_ca":false},{"name":"Wengao Gong","is_ca":false},{"name":"Charlene Tan","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09606259991720532,"gpt":0.5312202756607064,"spread":0.4351576757435011,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0143883,0.0004610327,0.0004786877,0.00114554,0.001546384,0.002997622,0.0008436659,0.000592549,0.0005372487],"category_scores_gemma":[0.06405678,0.0005116126,0.0002516494,0.0005428637,0.001540873,0.001893731,0.002702406,0.001143891,0.0001802078],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001599946,"about_ca_system_score_gemma":0.004630459,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00527504,"about_ca_topic_score_gemma":0.009447907,"domain_scores_codex":[0.9890084,0.006884721,0.000845956,0.0008798077,0.001803288,0.0005779425],"domain_scores_gemma":[0.9652471,0.02213945,0.003765853,0.001817177,0.005399879,0.001630621],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0002310929,0.001068498,0.1471467,0.0004743207,0.00002292046,0.0005006241,0.6324392,0.0008564424,0.02730686,0.001036797,0.0006278595,0.1882886],"study_design_scores_gemma":[0.0002305407,0.004367725,0.4733175,0.000723426,0.0001320912,0.001510987,0.4179944,0.01243173,0.04920613,0.003719695,0.03600125,0.0003645755],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9926892,0.00008420713,0.005154968,0.000158643,0.000006474003,0.0001702754,0.000008532467,0.00003949925,0.001688047],"genre_scores_gemma":[0.9873413,0.0001229962,0.01131608,0.00004581262,0.000004124359,0.0001371172,0.00001785815,0.00001479515,0.0009998094],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0143883,"threshold_uncertainty_score":0.07609349,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2122362059","doi":"10.1191/0265532206lt322oa","title":"Aiming for positive washback: a case study of international teaching assistants","year":2005,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":102,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université Laval","funders":"","keywords":"Test (biology); Psychology; Language proficiency; Process (computing); Mathematics education; Empirical research; Computer science","authors":[{"name":"Shahrzad Saif","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05418180584152639,"gpt":0.417552964658704,"spread":0.3633711588171776,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007090907,0.0009132704,0.0006940333,0.001177574,0.007476408,0.002904533,0.002595777,0.003783207,0.00242489],"category_scores_gemma":[0.02606587,0.0007124443,0.0006381013,0.0009941871,0.002688632,0.00161748,0.003125132,0.004107584,0.0005478734],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002406489,"about_ca_system_score_gemma":0.002727034,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004929525,"about_ca_topic_score_gemma":0.0130035,"domain_scores_codex":[0.9923446,0.004435326,0.0003190832,0.0004494324,0.000847509,0.001604164],"domain_scores_gemma":[0.9843876,0.008667849,0.001786283,0.0007516964,0.001098053,0.003308472],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0006586729,0.01561258,0.09269043,0.0005907764,0.00008675403,0.09080338,0.6864872,0.001088879,0.006645918,0.002685027,0.001792372,0.1008581],"study_design_scores_gemma":[0.000175311,0.006551826,0.04126441,0.000324124,0.00007951962,0.03349762,0.8831282,0.002379781,0.008789723,0.00160231,0.02205646,0.0001506972],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9960716,0.00008090044,0.001216433,0.0005325318,0.00001895767,0.000138884,0.00001137015,0.00001623256,0.001913105],"genre_scores_gemma":[0.992968,0.0002593973,0.003330395,0.0003558495,0.00003388179,0.0001489004,0.00001667783,0.00001965531,0.002867276],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007476408,"threshold_uncertainty_score":0.0375008,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2131806896","doi":"","title":"The Evolving Culture of Large-Scale Assessments in Canadian Education.","year":2008,"lang":"en","type":"article","venue":"","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":102,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Gatekeeping; Scale (ratio); Accountability; Christian ministry; Educational assessment; Political science; Medical education; Public relations; Geography; Psychology; Pedagogy; Medicine","authors":[{"name":"Don A. Klinger","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0193088662006192,"gpt":0.3692644063024438,"spread":0.3499555401018246,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02541595,0.000379494,0.000445602,0.003768843,0.02024308,0.01445344,0.003238576,0.0010117,0.001438835],"category_scores_gemma":[0.03550963,0.0005476752,0.0003171071,0.007164237,0.02122137,0.002374445,0.005853745,0.003246915,0.0001732277],"about_ca_system_candidate":true,"about_ca_system_consensus":true,"about_ca_system_score_codex":0.1637492,"about_ca_system_score_gemma":0.2795703,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.9857063,"about_ca_topic_score_gemma":0.9907783,"domain_scores_codex":[0.9663407,0.008796901,0.001589989,0.003010779,0.01703763,0.003224161],"domain_scores_gemma":[0.9161822,0.01947153,0.005553916,0.004185317,0.03866137,0.01594575],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001142115,0.0002190014,0.1085895,0.0004656782,0.00008066723,0.0008188799,0.6155456,0.002069137,0.003955861,0.03434943,0.01859028,0.2152017],"study_design_scores_gemma":[0.00002317451,0.0001104465,0.2848521,0.0007936919,0.00004544637,0.0004115462,0.4779721,0.002237227,0.00144669,0.006964155,0.2247454,0.0003979438],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8553696,0.003959606,0.0151191,0.04499346,0.0003419201,0.0005466656,0.0006085807,0.0003954977,0.07866564],"genre_scores_gemma":[0.9897913,0.0008759468,0.00392047,0.001027199,0.00001445925,0.00005677689,0.00009731614,0.00004985219,0.004166703],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.974584,"threshold_uncertainty_score":0.9699324,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2090290510","doi":"10.3138/cmlr.64.1.009","title":"AFL Research in the L2 Classroom and Evidence of Usefulness: Taking Formative Assessment to the Next Level","year":2007,"lang":"en","type":"article","venue":"Canadian Modern Language Review/ La Revue canadienne des langues vivantes","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":102,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Formative assessment; Assessment for learning; Mathematics education; Psychology; Curriculum; Pedagogy; Bridge (graph theory)","authors":[{"name":"Christian Colby‐Kelly","is_ca":false},{"name":"Carolyn E. Turner","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1996400635777524,"gpt":0.4220425672263123,"spread":0.2224025036485599,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.09745892,0.0006168346,0.0008266872,0.004193412,0.001897469,0.005959557,0.001790099,0.0009516652,0.001782733],"category_scores_gemma":[0.2059403,0.0002966829,0.0003552806,0.003262552,0.003856168,0.005079907,0.003978999,0.00146594,0.0004124207],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003134509,"about_ca_system_score_gemma":0.004569583,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007812535,"about_ca_topic_score_gemma":0.01019794,"domain_scores_codex":[0.9259647,0.0539064,0.003932798,0.002290107,0.01266599,0.001240105],"domain_scores_gemma":[0.7030655,0.2110507,0.02025037,0.01200722,0.05020304,0.003423191],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0004230235,0.001071295,0.3310831,0.001930481,0.000120585,0.0004868259,0.252088,0.0003687867,0.004233129,0.00268626,0.002071491,0.403437],"study_design_scores_gemma":[0.0001744285,0.008042185,0.6474378,0.005931977,0.0002199398,0.00210513,0.2592379,0.003193081,0.02097142,0.01047749,0.04187443,0.0003342583],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9733495,0.002847529,0.005606914,0.002330342,0.00009514637,0.0003049109,0.0001111184,0.00008875979,0.01526582],"genre_scores_gemma":[0.9946741,0.0006706693,0.003343377,0.000276395,0.00003907026,0.0001528668,0.00003803537,0.00001765014,0.0007878719],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.09745892,"threshold_uncertainty_score":0.5154182,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3202681787","doi":"10.1111/bjet.13169","title":"COVID‐19 as the tipping point for integrating e‐assessment in higher education practices","year":2021,"lang":"en","type":"article","venue":"British Journal of Educational Technology","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":102,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Sherbrooke","funders":"Université de Sherbrooke","keywords":"Tipping point (physics); Coronavirus disease 2019 (COVID-19); 2019-20 coronavirus outbreak; Severe acute respiratory syndrome coronavirus 2 (SARS-CoV-2); Higher education; Medicine; Virology; Economics; Engineering; Economic growth","authors":[{"name":"Christina St‐Onge","is_ca":true},{"name":"Kathleen Ouellet","is_ca":true},{"name":"Sawsen Lakhal","is_ca":true},{"name":"Tim Dubé","is_ca":true},{"name":"Mélanie Marceau","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0578890967993076,"gpt":0.4468809123947186,"spread":0.388991815595411,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05811482,0.0004332074,0.0006505613,0.001738585,0.01438787,0.0151485,0.002917985,0.00616033,0.005480587],"category_scores_gemma":[0.0688181,0.0006909842,0.0006591122,0.001572491,0.02842627,0.01434548,0.03445767,0.01092673,0.0006516923],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01364925,"about_ca_system_score_gemma":0.01866685,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004379637,"about_ca_topic_score_gemma":0.004353047,"domain_scores_codex":[0.8991832,0.07838005,0.003517239,0.002703794,0.01023999,0.005975689],"domain_scores_gemma":[0.9074454,0.05777471,0.007621267,0.005344026,0.009152693,0.01266189],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00007367213,0.0001876457,0.01082066,0.0004205479,0.00001471376,0.001908579,0.8759252,0.0004338141,0.001836347,0.05276906,0.004465989,0.05114376],"study_design_scores_gemma":[0.00001330725,0.0001589742,0.004679117,0.001336129,0.000006780589,0.0007747915,0.8623982,0.0008033165,0.0008040254,0.01737281,0.1115659,0.0000867246],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7248818,0.002539633,0.03879432,0.1423198,0.001098572,0.000550095,0.00005868317,0.000264629,0.08949249],"genre_scores_gemma":[0.9879007,0.000305297,0.006508812,0.003069345,0.00005537557,0.00009012926,0.00001487359,0.00003253478,0.002022827],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05811482,"threshold_uncertainty_score":0.3073443,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1996728665","doi":"10.1080/08878730.2012.760024","title":"Pedagogies for Preservice Assessment Education: Supporting Teacher Candidates' Assessment Literacy Development","year":2013,"lang":"en","type":"article","venue":"The Teacher Educator","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":95,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Praxis; Accountability; Teacher education; Literacy; Pedagogy; Psychology; Mathematics education; Perspective (graphical); Political science","authors":[{"name":"Christopher DeLuca","is_ca":true},{"name":"Teresa Rodríguez Chávez","is_ca":false},{"name":"Aarti P. Bellara","is_ca":false},{"name":"Chunhua Cao","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0388884580724605,"gpt":0.4471513230062795,"spread":0.408262864933819,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008363344,0.0003485318,0.0003043512,0.001150601,0.002785122,0.003332525,0.0009027171,0.0008422112,0.003076542],"category_scores_gemma":[0.02463933,0.0002510104,0.0002543079,0.0005563675,0.00120221,0.001488898,0.004480828,0.001933481,0.001001555],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001396059,"about_ca_system_score_gemma":0.009341176,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007618944,"about_ca_topic_score_gemma":0.00362513,"domain_scores_codex":[0.9946302,0.00366566,0.0002269226,0.0002321697,0.000854563,0.0003904557],"domain_scores_gemma":[0.9848215,0.006857314,0.002279341,0.0007831948,0.001697649,0.00356109],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001223903,0.002974879,0.07727726,0.0008698273,0.0000138437,0.001085346,0.1918638,0.0005455047,0.008623611,0.009879582,0.01218521,0.6945589],"study_design_scores_gemma":[0.0002012323,0.004513059,0.2211301,0.003864583,0.0001046034,0.005087996,0.4261155,0.005016373,0.02126547,0.0267928,0.285707,0.0002011296],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8770329,0.001310107,0.04041338,0.01862984,0.0002545872,0.001125317,0.00005716383,0.0007775932,0.06039909],"genre_scores_gemma":[0.956622,0.0006136704,0.03640751,0.000503625,0.00004515772,0.0005525515,0.00003408072,0.00002855807,0.005192851],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008363344,"threshold_uncertainty_score":0.04423016,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2906729407","doi":"10.5539/ies.v12n1p61","title":"Application of Rubrics in the Classroom: A Vital Tool for Improvement in Assessment, Feedback and Learning","year":2018,"lang":"en","type":"article","venue":"International Education Studies","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":89,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Rubric; Grading (engineering); Strengths and weaknesses; Mathematics education; Computer science; Peer assessment; Standards-based assessment; Teaching method; Psychology; Educational assessment; Engineering","authors":[{"name":"Faieza Chowdhury","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0463357328462515,"gpt":0.449002478674172,"spread":0.4026667458279205,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04194926,0.001702849,0.001833211,0.01106071,0.002013546,0.007017651,0.003142834,0.001806821,0.007484126],"category_scores_gemma":[0.141868,0.0007836456,0.0007540383,0.006546521,0.002997244,0.007692509,0.00443981,0.004416753,0.008593018],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001637149,"about_ca_system_score_gemma":0.005469329,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001731794,"about_ca_topic_score_gemma":0.002343722,"domain_scores_codex":[0.9385698,0.02503946,0.004822421,0.002080232,0.02882693,0.0006612387],"domain_scores_gemma":[0.8485611,0.06414711,0.01613017,0.01936846,0.04751414,0.004278972],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.00005149847,0.000235544,0.001692944,0.001078605,0.00002340682,0.00009212432,0.003350614,0.000412075,0.00751084,0.004302811,0.05059117,0.9306585],"study_design_scores_gemma":[0.0001335002,0.00151558,0.02736456,0.004699061,0.00009303227,0.002183113,0.00552539,0.008571168,0.02390085,0.02449282,0.9009984,0.0005224897],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0321283,0.01410077,0.8036759,0.01719566,0.003204002,0.003709858,0.001181918,0.0458534,0.07895018],"genre_scores_gemma":[0.0758265,0.005992949,0.8933443,0.001341373,0.0009272196,0.001421161,0.001186082,0.002621678,0.01733878],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04194926,"threshold_uncertainty_score":0.2218515,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2107226435","doi":"10.6018/ijes.13.2.185891","title":"Washback in language assessment","year":2013,"lang":"en","type":"article","venue":"International Journal of English Studies","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":88,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Test (biology); Psychology; Quarter (Canadian coin); Mathematics education; Medical education; Pedagogy; Medicine; Geography","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.02948544944097654,"gpt":0.421665550082896,"spread":0.3921801006419194,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02226993,0.0007997573,0.001049478,0.002515795,0.001784158,0.005966575,0.002069131,0.002657767,0.009900821],"category_scores_gemma":[0.08076338,0.0004546521,0.0005013522,0.001381838,0.004590767,0.01010772,0.008163041,0.003370399,0.002437495],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002822267,"about_ca_system_score_gemma":0.002639923,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001354042,"about_ca_topic_score_gemma":0.001661362,"domain_scores_codex":[0.9686812,0.0206864,0.001452608,0.001325398,0.007129975,0.0007244766],"domain_scores_gemma":[0.9403295,0.04574199,0.002606587,0.003546007,0.006667883,0.001108007],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003682147,0.0003732464,0.003637181,0.00154276,0.00002841271,0.000266989,0.01655864,0.0003755518,0.001850966,0.02907296,0.005816729,0.9401084],"study_design_scores_gemma":[0.00024282,0.003369192,0.02228644,0.01334107,0.0001760918,0.004561818,0.04479221,0.003655473,0.01421184,0.2230545,0.6699219,0.0003865726],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.2411229,0.1540432,0.2685151,0.04781415,0.00668704,0.001402615,0.0001715024,0.00235295,0.2778906],"genre_scores_gemma":[0.7940065,0.02925789,0.1167138,0.01020365,0.001004702,0.00100485,0.0001437557,0.0006390071,0.0470259],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02226993,"threshold_uncertainty_score":0.1177761,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2781958120","doi":"10.1016/j.tate.2017.12.010","title":"Changing approaches to classroom assessment: An empirical study across teacher career stages","year":2018,"lang":"en","type":"article","venue":"Teaching and Teacher Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":87,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Set (abstract data type); Psychology; Mathematics education; Empirical research; Pedagogy; Computer science","authors":[{"name":"Andrew Coombs","is_ca":true},{"name":"Christopher DeLuca","is_ca":true},{"name":"Danielle LaPointe-McEwan","is_ca":true},{"name":"Agnieszka Chalas","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.159263218109638,"gpt":0.4565673728995994,"spread":0.2973041547899614,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02136839,0.000220009,0.0004909112,0.002179091,0.005154816,0.004270278,0.001729722,0.001361095,0.001552006],"category_scores_gemma":[0.1138471,0.0005518003,0.0004703171,0.001872732,0.002685763,0.002574763,0.004578011,0.002650517,0.0002885124],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004399624,"about_ca_system_score_gemma":0.007464694,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01182829,"about_ca_topic_score_gemma":0.03217481,"domain_scores_codex":[0.9819115,0.01142591,0.0009356186,0.001138606,0.003073115,0.00151527],"domain_scores_gemma":[0.869696,0.0901467,0.01074771,0.004740595,0.01730058,0.007368472],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0005833859,0.00302197,0.5414973,0.0001813684,0.00004587229,0.0003625768,0.3789269,0.0002433894,0.002265548,0.0014592,0.0003651216,0.0710474],"study_design_scores_gemma":[0.00004501277,0.001050794,0.6825288,0.0001966715,0.00003948124,0.0003421177,0.3077647,0.0005959724,0.001076688,0.0006829583,0.005621598,0.00005526859],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9974706,0.0001403261,0.0003601115,0.0001488928,0.000006743544,0.00005992999,0.00001529977,0.000004385192,0.001793693],"genre_scores_gemma":[0.9983346,0.00008866643,0.0006802656,0.00004977879,0.000002132701,0.00007556881,0.00001792053,0.000005073457,0.000745985],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02136839,"threshold_uncertainty_score":0.1130081,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1599897745","doi":"10.18806/tesl.v23i2.53","title":"Effects of Peer Feedback on EFL Student Writers at Different Levels of English Proficiency: A Japanese Context","year":2006,"lang":"en","type":"article","venue":"TESL Canada Journal","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":84,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Peer feedback; Psychology; Context (archaeology); Peer evaluation; Mathematics education; English as a foreign language; Peer group; Higher education; Pedagogy; Social psychology","authors":[{"name":"Taeko Kamimura","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01525476707009359,"gpt":0.284865287494137,"spread":0.2696105204240434,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006329191,0.0006594701,0.0008055572,0.0008242865,0.001607104,0.001128609,0.0005208883,0.0007043291,0.001337506],"category_scores_gemma":[0.06671613,0.000238562,0.0002886353,0.0003846157,0.0007999227,0.0006970274,0.001739477,0.0005647016,0.0002028709],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000468222,"about_ca_system_score_gemma":0.0008219139,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001482534,"about_ca_topic_score_gemma":0.00261132,"domain_scores_codex":[0.9868271,0.009353414,0.0005949377,0.0006810378,0.002064298,0.0004791586],"domain_scores_gemma":[0.928274,0.04940344,0.00740787,0.002277199,0.007943152,0.004694289],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.008158998,0.008970548,0.3985783,0.001136069,0.0003789988,0.003080223,0.1682367,0.0006582762,0.09218762,0.0002191898,0.000825502,0.3175696],"study_design_scores_gemma":[0.0006187594,0.03681539,0.8163277,0.0003012247,0.0005255767,0.001745023,0.09663276,0.002047351,0.03943862,0.0005327021,0.004847407,0.000167469],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9992812,0.0000744361,0.0001547438,0.00003093326,0.000006095781,0.00002682308,0.000005299764,0.000008965833,0.0004113974],"genre_scores_gemma":[0.9987701,0.00008931851,0.0006620936,0.00002363376,0.0000135718,0.00003924435,0.00001170601,0.000005302148,0.0003850612],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006329191,"threshold_uncertainty_score":0.03347236,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2461301006","doi":"10.5539/elt.v9n8p106","title":"Effectiveness of Using Screencast Feedback on EFL Students’ Writing and Perception","year":2016,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":83,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Psychology; Perception; Peer feedback; Constructive; Video feedback; Control (management); Mathematics education; Pedagogy; Computer science","authors":[{"name":"Amira Desouky Ali","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01930918518031864,"gpt":0.3550008529784101,"spread":0.3356916677980915,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005466695,0.000395678,0.0006055774,0.0005722427,0.0004792664,0.001190397,0.000429625,0.0005191744,0.002386369],"category_scores_gemma":[0.02058085,0.0001742957,0.0003426926,0.0002954209,0.0004012136,0.000540288,0.0006189092,0.0005292973,0.0003181812],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004147885,"about_ca_system_score_gemma":0.0006041238,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000796042,"about_ca_topic_score_gemma":0.001120861,"domain_scores_codex":[0.9967917,0.001527748,0.0003080529,0.0002953742,0.0008405485,0.0002366514],"domain_scores_gemma":[0.9770277,0.01516428,0.003655724,0.0009321452,0.001878884,0.001341355],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.008812625,0.02385774,0.3474854,0.001939057,0.0002327001,0.0005745931,0.06611079,0.0007900888,0.08151216,0.0001968642,0.0008975406,0.4675905],"study_design_scores_gemma":[0.0004829484,0.04533347,0.8812281,0.0003254042,0.000342512,0.0002492398,0.03468902,0.001857422,0.03159574,0.0002791689,0.003496944,0.0001198942],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9992873,0.00003388926,0.0001862605,0.00002456598,0.000006910037,0.00004533448,0.00001510186,0.00001059767,0.0003901818],"genre_scores_gemma":[0.9981095,0.0000658645,0.0009843453,0.00002938477,0.0000110977,0.00009781415,0.00002832808,0.000003775431,0.0006700167],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005466695,"threshold_uncertainty_score":0.02891099,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1975314479","doi":"10.3138/cmlr.64.1.199","title":"Assessment for Learning: Integrating Assessment, Teaching, and Learning in the ESL/EFL Writing Classroom","year":2007,"lang":"en","type":"article","venue":"Canadian Modern Language Review/ La Revue canadienne des langues vivantes","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":82,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Assessment for learning; Mathematics education; Writing assessment; Pedagogy; Psychology; Computer science; Formative assessment","authors":[{"name":"Icy Lee","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02022232402502297,"gpt":0.33829144772885,"spread":0.318069123703827,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0234808,0.0004143289,0.0006987352,0.00274049,0.001528626,0.004809333,0.001379587,0.0009244198,0.00137285],"category_scores_gemma":[0.03528127,0.0001962291,0.0002568568,0.002588273,0.003777957,0.003171573,0.003644067,0.001773697,0.0002567907],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008201598,"about_ca_system_score_gemma":0.02515505,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03631164,"about_ca_topic_score_gemma":0.04681213,"domain_scores_codex":[0.9829616,0.01184323,0.0005841579,0.0004600078,0.003820014,0.0003310303],"domain_scores_gemma":[0.9830284,0.009968736,0.001213924,0.0005984851,0.00416334,0.001027023],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00004931093,0.0001959424,0.006432895,0.001267978,0.00001971284,0.0001885041,0.01460687,0.0004374424,0.0009096339,0.02696107,0.008010125,0.9409206],"study_design_scores_gemma":[0.0002916579,0.001407069,0.1093688,0.01137934,0.000188411,0.002503436,0.05855072,0.01051132,0.007850726,0.1337949,0.6638736,0.0002800551],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.27801,0.1263166,0.2116971,0.08280223,0.002714295,0.002217733,0.0002317632,0.002246402,0.2937638],"genre_scores_gemma":[0.8141385,0.01711079,0.1572776,0.001758642,0.0002599586,0.0006010736,0.00007220733,0.00007582657,0.008705395],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03631164,"threshold_uncertainty_score":0.1241798,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2782375796","doi":"10.1016/j.stueduc.2017.12.008","title":"Re-conceptualizing classroom assessment fairness: A systematic meta-ethnography of assessment literature and beyond","year":2018,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":82,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Conceptualization; Accountability; Construct (python library); Educational assessment; Psychology; Ethnography; Standards-based assessment; Process (computing); Pedagogy; Alternative assessment; Mathematics education; Sociology; Computer science; Political science","authors":[{"name":"Amirhossein Rasooli","is_ca":false},{"name":"Hamed Zandi","is_ca":false},{"name":"Christopher DeLuca","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1824586292219303,"gpt":0.5132713437504184,"spread":0.330812714528488,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.277291,0.001611825,0.004838464,0.01287788,0.002855372,0.009569612,0.004006105,0.002162059,0.001645115],"category_scores_gemma":[0.4587441,0.001602621,0.003636483,0.008381753,0.006940768,0.0172155,0.008853901,0.005189211,0.0001038851],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009211442,"about_ca_system_score_gemma":0.02547554,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01019809,"about_ca_topic_score_gemma":0.01974306,"domain_scores_codex":[0.7444218,0.2075059,0.02722476,0.00859302,0.01034966,0.001904906],"domain_scores_gemma":[0.2614102,0.6871106,0.01856062,0.01928192,0.01277231,0.0008642738],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0007275234,0.0003338827,0.0280406,0.1276552,0.01351476,0.000414186,0.4336383,0.001481132,0.001736386,0.02693461,0.003276421,0.3622469],"study_design_scores_gemma":[0.0005584494,0.0009179476,0.03419867,0.4212689,0.03022719,0.0009804715,0.3814472,0.003945926,0.004883492,0.06634191,0.05475406,0.0004758645],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.3237903,0.5335526,0.1101427,0.01736415,0.001552993,0.006101578,0.00087982,0.0001562077,0.006459626],"genre_scores_gemma":[0.8689556,0.05447753,0.06654858,0.00378761,0.0001951551,0.005161455,0.0003582157,0.0001194692,0.0003963358],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.722709,"threshold_uncertainty_score":0.8912289,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2140654065","doi":"","title":"Formative Assessment and the Contemporary Classroom: Synergies and Tensions between Research and Practice.","year":2011,"lang":"en","type":"article","venue":"Canadian Journal of Education / Revue canadienne de l éducation","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":82,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":true},"ca_institutions":"Brock University","funders":"","keywords":"Formative assessment; Variety (cybernetics); Professional development; Pedagogy; Mathematics education; Psychology; Faculty development; Teacher education; Medical education; Sociology; Medicine; Computer science","authors":[{"name":"Louis Volante","is_ca":true},{"name":"Danielle Beckett","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2594907877898358,"gpt":0.4333496664294458,"spread":0.17385887863961,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06078527,0.0003098734,0.000396883,0.003512128,0.004047755,0.009527551,0.002029655,0.001672057,0.0007184774],"category_scores_gemma":[0.1636944,0.0003712462,0.0001375265,0.004539957,0.02073988,0.005822222,0.003959887,0.002222047,0.00007618083],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01657455,"about_ca_system_score_gemma":0.02701417,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.06780478,"about_ca_topic_score_gemma":0.1377289,"domain_scores_codex":[0.9528635,0.03224491,0.001887572,0.001244437,0.0105816,0.001178085],"domain_scores_gemma":[0.8012773,0.1696208,0.008312603,0.004568855,0.01305141,0.003169016],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.00004716161,0.0001566351,0.02167033,0.0006638361,0.00002130254,0.0002407453,0.6889727,0.0001158481,0.000641271,0.01785862,0.001533473,0.2680781],"study_design_scores_gemma":[0.00003913403,0.0002774477,0.08533072,0.003445296,0.00004440866,0.00103601,0.7886438,0.0006115046,0.001139443,0.03063143,0.08872092,0.00007986328],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7948279,0.07866579,0.02743819,0.05789084,0.0006268189,0.0005268023,0.00008172711,0.0001156361,0.03982628],"genre_scores_gemma":[0.9775637,0.01030012,0.008453204,0.001485628,0.0001278386,0.0002883403,0.00002347594,0.00001380458,0.001743834],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06780478,"threshold_uncertainty_score":0.321467,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2167507251","doi":"10.1111/medu.12517","title":"Automated essay scoring and the future of educational assessment in medical education","year":2014,"lang":"en","type":"article","venue":"Medical Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":80,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Medical Council of Canada; University of Alberta","funders":"","keywords":"Summative assessment; Formative assessment; Computer science; Process (computing); Context (archaeology); Scoring system; Writing assessment; Educational measurement; Artificial intelligence; Natural language processing; Machine learning; Mathematics education; Curriculum; Psychology; Medicine; Pedagogy","authors":[{"name":"Mark J. Gierl","is_ca":true},{"name":"Syed Latifi","is_ca":true},{"name":"Hollis Lai","is_ca":true},{"name":"André-Philippe Boulais","is_ca":true},{"name":"André De Champlain","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.007292918176272418,"gpt":0.3889499880168408,"spread":0.3816570698405684,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05951421,0.001398052,0.001239081,0.006989547,0.0008061093,0.007054998,0.003336206,0.003515362,0.00488366],"category_scores_gemma":[0.1945858,0.0004612153,0.0006371727,0.004369089,0.003251604,0.007326131,0.002921549,0.002574365,0.002630769],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002284631,"about_ca_system_score_gemma":0.003589103,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002636017,"about_ca_topic_score_gemma":0.002160897,"domain_scores_codex":[0.9566417,0.03041275,0.002153307,0.001925783,0.008348688,0.0005177679],"domain_scores_gemma":[0.7734265,0.1412965,0.01515012,0.01038767,0.05477065,0.004968605],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002079735,0.0001981857,0.01445109,0.0005183487,0.0000601301,0.00003717383,0.0004447543,0.004242156,0.0009994615,0.008236101,0.01182421,0.9587803],"study_design_scores_gemma":[0.0006179785,0.002913273,0.1575929,0.006719451,0.0002851839,0.001467827,0.003072107,0.307238,0.008224366,0.2928403,0.2181267,0.0009019316],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1134043,0.1210997,0.6425041,0.06648628,0.004146755,0.001356276,0.00147103,0.008384572,0.04114687],"genre_scores_gemma":[0.4219981,0.02143787,0.5429711,0.003127821,0.003064231,0.00116636,0.001140358,0.0003732296,0.004720897],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.05951421,"threshold_uncertainty_score":0.314745,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2086922103","doi":"10.1002/tesq.105","title":"Motivation and Test Anxiety in Test Performance Across Three Testing Contexts: The <scp>CAEL</scp>,<scp> CET</scp>, and <scp>GEPT</scp>","year":2013,"lang":"en","type":"article","venue":"TESOL Quarterly","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":77,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"Mount Saint Vincent University; Carleton University; Queen's University","funders":"U.S. Department of Energy","keywords":"Test anxiety; Test (biology); Psychology; Context (archaeology); Anxiety; Social psychology; Language assessment; Developmental psychology; Mathematics education","authors":[{"name":"Liying Cheng","is_ca":true},{"name":"Don A. Klinger","is_ca":true},{"name":"Janna Fox","is_ca":true},{"name":"Christine Doe","is_ca":true},{"name":"Yan Jin","is_ca":false},{"name":"Jessica R. W. Wu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02374035340395477,"gpt":0.2807221828600872,"spread":0.2569818294561324,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002809433,0.0002707127,0.0002411286,0.001185012,0.0008142036,0.001718232,0.0003532097,0.0003466091,0.0005833157],"category_scores_gemma":[0.0149025,0.0001499974,0.0003877413,0.0007387298,0.0009554401,0.0005271804,0.00148468,0.0007916858,0.00007873195],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001351067,"about_ca_system_score_gemma":0.001303028,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02078827,"about_ca_topic_score_gemma":0.03672275,"domain_scores_codex":[0.9969485,0.001050869,0.000179002,0.0002326798,0.001165052,0.0004239567],"domain_scores_gemma":[0.9883909,0.003825821,0.004290421,0.0003146394,0.001254299,0.001924061],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00004681199,0.0002414072,0.9902817,0.000009878059,0.00003478831,0.0000434621,0.002602399,0.0000505903,0.0003473089,0.00007044752,0.00005598223,0.006215188],"study_design_scores_gemma":[0.000001450096,0.00008405945,0.9979073,0.000004998489,0.000006651118,0.00002298515,0.001681491,0.00008841397,0.0001011309,0.00002282336,0.00007344694,0.000005379879],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.999671,0.00002410716,0.00002587814,0.00002104351,0.00000113472,0.000002906428,0.000005977541,9.931804e-7,0.0002469325],"genre_scores_gemma":[0.9998809,0.0000143495,0.0000309995,0.000008856715,0.00000138117,0.000003743177,0.00001388787,6.31728e-7,0.00004520512],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02078827,"threshold_uncertainty_score":0.04133457,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2108440479","doi":"10.1080/00131725.2011.577669","title":"Being Fair: Teachers’ Interpretations of Principles for Standards-Based Grading","year":2011,"lang":"en","type":"article","venue":"The Educational Forum","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":77,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Grading (engineering); Mathematics education; Psychology; Academic standards; Set (abstract data type); Pedagogy; Higher education; Computer science; Engineering; Political science","authors":[{"name":"Robin D. Tierney","is_ca":true},{"name":"Marielle Simon","is_ca":true},{"name":"Julie Charland","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07001771760809876,"gpt":0.3902159675238942,"spread":0.3201982499157954,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1389061,0.0004608569,0.0007692131,0.003779833,0.005635107,0.01273337,0.00304325,0.003933853,0.001159255],"category_scores_gemma":[0.2755595,0.0007734571,0.0008264262,0.001660188,0.03366331,0.00790163,0.007413887,0.01100981,0.0002206531],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007445518,"about_ca_system_score_gemma":0.007480277,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004296388,"about_ca_topic_score_gemma":0.005411205,"domain_scores_codex":[0.8209336,0.1313624,0.01004125,0.0050316,0.02934013,0.003290981],"domain_scores_gemma":[0.6819415,0.2187569,0.02683638,0.02397194,0.04268403,0.005809252],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001760076,0.0002725214,0.02042395,0.0003215596,0.00006809943,0.0003377488,0.4660221,0.002181649,0.001930434,0.4616282,0.005198197,0.04143958],"study_design_scores_gemma":[0.0001296184,0.0002491492,0.02208037,0.001030688,0.00007268671,0.000388578,0.1226875,0.01443834,0.004188368,0.7757692,0.05868952,0.0002760997],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5465029,0.00151015,0.2734479,0.07963615,0.001638525,0.0004933371,0.0001068447,0.0003695308,0.09629463],"genre_scores_gemma":[0.9840046,0.00008894556,0.01424993,0.0007320755,0.00005536276,0.00008371798,0.00001194175,0.00003839704,0.000735063],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1389061,"threshold_uncertainty_score":0.7346146,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2600733185","doi":"10.1080/0969594x.2017.1297010","title":"Developing assessment capable teachers in this age of accountability","year":2017,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":75,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Accountability; Psychology; Medical education; Political science; Medicine","authors":[{"name":"Christopher DeLuca","is_ca":true},{"name":"Sandra Johnson","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1225845974022681,"gpt":0.5148076355503635,"spread":0.3922230381480953,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02269946,0.0003747622,0.0005151991,0.0008956062,0.006982223,0.01347385,0.001655869,0.00627538,0.009376938],"category_scores_gemma":[0.04361229,0.000553527,0.0002431988,0.000656278,0.007994081,0.01824725,0.01673678,0.01058218,0.005695342],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004101363,"about_ca_system_score_gemma":0.02015616,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003459565,"about_ca_topic_score_gemma":0.007254483,"domain_scores_codex":[0.9839932,0.008023634,0.0006391404,0.001360573,0.003425587,0.002557921],"domain_scores_gemma":[0.9529876,0.01649808,0.0041022,0.003371358,0.009192582,0.0138483],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001354123,0.0005840354,0.01565807,0.0006884508,0.00002333394,0.001180015,0.1090157,0.001080033,0.003226972,0.4407865,0.1393012,0.2883203],"study_design_scores_gemma":[0.00003705346,0.0001625896,0.006482315,0.0009940208,0.00001204404,0.0008776822,0.0601386,0.001720745,0.002006557,0.1831348,0.7443578,0.00007589757],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.1104717,0.007725571,0.09586354,0.5193545,0.004392211,0.0003220362,0.0001192629,0.001345703,0.2604055],"genre_scores_gemma":[0.7978379,0.006200644,0.05171781,0.0419795,0.001152595,0.0005111528,0.0001518846,0.0004090033,0.1000395],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02269946,"threshold_uncertainty_score":0.1200476,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1996435010","doi":"10.1080/09695940701272773","title":"Did we take the same test? Differing accounts of the Ontario Secondary School Literacy Test by first and second language test‐takers","year":2007,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":74,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"Queen's University; Carleton University","funders":"","keywords":"Test (biology); Graduation (instrument); Psychology; Context (archaeology); Construct (python library); Literacy; Construct validity; Mathematics education; Fidelity; Test score; Standardized test; Pedagogy; Developmental psychology; Computer science; Psychometrics; Mathematics","authors":[{"name":"Janna Fox","is_ca":true},{"name":"Liying Cheng","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02100229501179931,"gpt":0.3845813286896781,"spread":0.3635790336778789,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01524126,0.0004802489,0.0004578446,0.003077692,0.006924826,0.006276983,0.002250515,0.001682204,0.001397538],"category_scores_gemma":[0.06272691,0.0004507829,0.0006178707,0.002554065,0.01692384,0.003919709,0.004454658,0.002445285,0.0002928285],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.03391194,"about_ca_system_score_gemma":0.01786655,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.7256564,"about_ca_topic_score_gemma":0.7678544,"domain_scores_codex":[0.9688408,0.01145943,0.001428998,0.00179725,0.0143978,0.00207573],"domain_scores_gemma":[0.9702678,0.0119373,0.005158749,0.002780956,0.007800633,0.002054555],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.00008083533,0.00004395532,0.09512686,0.0001115506,0.00003332541,0.0006355554,0.8630092,0.0001099924,0.001013146,0.0153206,0.002556098,0.02195892],"study_design_scores_gemma":[0.000029515,0.0001821476,0.3444324,0.0006555044,0.00007416746,0.001019792,0.5689775,0.001114598,0.001566736,0.009352142,0.07239216,0.0002033274],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9258767,0.00203418,0.004057247,0.01825272,0.0001374917,0.0001566751,0.0003545991,0.00005569515,0.04907459],"genre_scores_gemma":[0.9910364,0.0006404976,0.001204713,0.001426436,0.00003541164,0.00005878621,0.0001457196,0.00005185678,0.005400038],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7256564,"threshold_uncertainty_score":0.5519184,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4400440977","doi":"10.55016/ojs/ajer.v53i1.55195","title":"Factors Affecting Teachers’ Grading and Assessment Practices","year":2007,"lang":"en","type":"article","venue":"Alberta Journal of Educational Research","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":73,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Grading (engineering); Psychology; Mathematics education; Pedagogy; Engineering","authors":[{"name":"Caitlin Duncan","is_ca":false},{"name":"Brian Noonan","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2729481972337564,"gpt":0.5720544857102489,"spread":0.2991062884764925,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00809396,0.000176987,0.0003085039,0.001074199,0.0006850237,0.001302738,0.0004755265,0.00035361,0.001053938],"category_scores_gemma":[0.07253569,0.0002303415,0.0002076323,0.001306556,0.0008566346,0.0007118857,0.0005409879,0.0005050994,0.0002670718],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001179464,"about_ca_system_score_gemma":0.001387941,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01523382,"about_ca_topic_score_gemma":0.02152471,"domain_scores_codex":[0.98625,0.005641961,0.00183303,0.001233484,0.004397418,0.000644091],"domain_scores_gemma":[0.8966005,0.06219422,0.02392231,0.004010211,0.009567021,0.003705782],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000138788,0.0001773447,0.9575471,0.00006898767,0.0000456347,0.0001605286,0.007043403,0.0004513339,0.001572106,0.0001069128,0.000342861,0.03234495],"study_design_scores_gemma":[0.000008521441,0.0001216378,0.995652,0.00001975924,0.00001317393,0.0001188242,0.00231199,0.0003167185,0.0004869119,0.0000974345,0.0008411086,0.00001187739],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9983188,0.0001562511,0.0003698596,0.0001227918,0.000005912064,0.0000188004,0.00002929676,0.00001119584,0.0009670118],"genre_scores_gemma":[0.9994659,0.00005096152,0.0002591497,0.00001208426,0.000004310551,0.000007523466,0.00002509202,0.000002871758,0.0001722138],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01523382,"threshold_uncertainty_score":0.04280543,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2013464569","doi":"10.1080/0969594x.2013.776943","title":"Fair and equitable assessment practices for all students","year":2013,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":71,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Thompson Rivers University; University of Lethbridge; University of Alberta; University of Calgary","funders":"","keywords":"Intrusiveness; Equity (law); Psychology; Public relations; Best practice; Medical education; Applied psychology; Pedagogy; Social psychology; Political science; Medicine","authors":[{"name":"Shelleyann Scott","is_ca":true},{"name":"Charles F. Webber","is_ca":true},{"name":"Judy Lupart","is_ca":true},{"name":"Nola Aitken","is_ca":true},{"name":"Donald E. Scott","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1023168743994288,"gpt":0.5279200270248844,"spread":0.4256031526254556,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06826248,0.0004899547,0.000748409,0.003049268,0.007909843,0.008764635,0.002935407,0.00293724,0.003494638],"category_scores_gemma":[0.1862538,0.0003797446,0.0004884062,0.002153192,0.003962099,0.007811042,0.01270242,0.003920188,0.001094227],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004941695,"about_ca_system_score_gemma":0.01813309,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003867726,"about_ca_topic_score_gemma":0.008385749,"domain_scores_codex":[0.860747,0.09396933,0.00852835,0.004317063,0.02781846,0.004619803],"domain_scores_gemma":[0.8434201,0.04804323,0.0169552,0.03193507,0.04756581,0.01208055],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001590673,0.001263416,0.07330026,0.000347795,0.00004293047,0.0002875899,0.08340397,0.002118492,0.00285941,0.04572968,0.01537773,0.7751097],"study_design_scores_gemma":[0.0001616588,0.002252053,0.1741936,0.004081983,0.0000940471,0.002019597,0.1844827,0.01030888,0.0157739,0.2698782,0.3360912,0.0006621159],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6514493,0.001850122,0.1163223,0.07695903,0.0006635918,0.002417034,0.0001831516,0.001277564,0.1488779],"genre_scores_gemma":[0.9596473,0.0002847927,0.03225647,0.001603969,0.00006321611,0.0003984181,0.00004394733,0.00004973496,0.005652059],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06826248,"threshold_uncertainty_score":0.3610108,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2923740196","doi":"10.1080/0969594x.2019.1593105","title":"Conceptualising fairness in classroom assessment: exploring the value of organisational justice theory","year":2019,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":71,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Foundation (evidence); Economic Justice; Scholarship; Value (mathematics); Core (optical fiber); Sociology; Psychology; Social psychology; Epistemology; Political science; Computer science; Law","authors":[{"name":"Amirhossein Rasooli","is_ca":true},{"name":"Hamed Zandi","is_ca":false},{"name":"Christopher DeLuca","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07905727411628827,"gpt":0.4421515498552321,"spread":0.3630942757389438,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05276362,0.0005648842,0.001095563,0.00401221,0.005696093,0.01490951,0.00298669,0.004119467,0.002540268],"category_scores_gemma":[0.09571839,0.0003929282,0.0007136008,0.002635494,0.03781327,0.01612487,0.01367879,0.00772952,0.0002230133],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01013531,"about_ca_system_score_gemma":0.01337134,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005827148,"about_ca_topic_score_gemma":0.005517413,"domain_scores_codex":[0.9291825,0.05777879,0.001687789,0.001897491,0.007637205,0.001816342],"domain_scores_gemma":[0.8795193,0.1027062,0.005209051,0.004364857,0.00583383,0.002366783],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002629136,0.0001162947,0.003882583,0.0002293749,0.00001936343,0.00007596194,0.03673689,0.001725155,0.0001490604,0.9147874,0.0005957686,0.04165575],"study_design_scores_gemma":[0.00001276687,0.00004763998,0.001803405,0.0006705945,0.00001390339,0.0000785082,0.01551375,0.004920866,0.0002148219,0.9650324,0.0116607,0.00003067347],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1768838,0.01472522,0.565375,0.0845071,0.001021756,0.0004018998,0.00005189651,0.000112342,0.1569209],"genre_scores_gemma":[0.9659733,0.001263975,0.03027358,0.00113332,0.0001637282,0.0001503856,0.00000934343,0.00002116422,0.001011118],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.05276362,"threshold_uncertainty_score":0.279044,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1996477132","doi":"10.1177/0741088310371635","title":"Undergraduate Writing Assignments: An Analysis of Syllabi at One Canadian College","year":2010,"lang":"en","type":"article","venue":"Written Communication","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":69,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"Western University; University of Alberta","funders":"","keywords":"Syllabus; Mathematics education; Curriculum; Higher education; Task (project management); Academic writing; Psychology; Computer science; Pedagogy; Engineering","authors":[{"name":"Roger Graves","is_ca":true},{"name":"Theresa Hyland","is_ca":false},{"name":"Boba Samuels","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03217846521966579,"gpt":0.3400945437894142,"spread":0.3079160785697485,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001851715,0.0003466083,0.000279731,0.007493525,0.003223037,0.001354932,0.0005976873,0.0002571035,0.00169119],"category_scores_gemma":[0.02025422,0.0002413759,0.0001906113,0.006624245,0.0007710853,0.0003160977,0.001043775,0.0004186003,0.0004448985],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006787115,"about_ca_system_score_gemma":0.01198891,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.4844484,"about_ca_topic_score_gemma":0.7001063,"domain_scores_codex":[0.99696,0.0003189514,0.0001882463,0.0002914949,0.001708801,0.0005324847],"domain_scores_gemma":[0.9784333,0.003722897,0.002749569,0.0005161067,0.01138905,0.003189171],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"qualitative","study_design_scores_codex":[0.0003908891,0.0003860544,0.6916221,0.0002450222,0.00003330124,0.0007819294,0.05479982,0.0003174714,0.01063873,0.0005373938,0.002924063,0.2373232],"study_design_scores_gemma":[0.000004335132,0.00007394904,0.9859586,0.00002624987,0.000007391184,0.0001020289,0.007100414,0.0004246854,0.001268726,0.00005470755,0.004954615,0.00002428022],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9963057,0.00007719718,0.0002697043,0.00004894892,0.000008505333,0.00009287864,0.0004212201,0.00002428688,0.002751575],"genre_scores_gemma":[0.9936825,0.0001683048,0.001427342,0.00003683141,0.000008053629,0.00005114394,0.001009578,0.00002127212,0.003594979],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5155516,"threshold_uncertainty_score":0.9632572,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4388945822","doi":"10.3389/feduc.2023.1270700","title":"Challenges and opportunities for classroom-based formative assessment and AI: a perspective article","year":2023,"lang":"en","type":"article","venue":"Frontiers in Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":65,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Kellogg's (Canada)","funders":"","keywords":"Formative assessment; Perspective (graphical); Computer science; Mathematics education; Engineering ethics; Psychology; Artificial intelligence; Engineering","authors":[{"name":"Therese N. Hopfenbeck","is_ca":true},{"name":"Zhonghua Zhang","is_ca":false},{"name":"Sundance Zhihong Sun","is_ca":false},{"name":"Pamela Robertson","is_ca":false},{"name":"Joshua A. McGrane","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08248159969855182,"gpt":0.404538068375617,"spread":0.3220564686770652,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1412789,0.000896756,0.001183955,0.0035651,0.00443914,0.02446334,0.003992936,0.006295463,0.002505542],"category_scores_gemma":[0.1479911,0.0006547209,0.0007612188,0.003099733,0.02184929,0.02397939,0.01111995,0.00909814,0.000787398],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006932844,"about_ca_system_score_gemma":0.01642801,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002912948,"about_ca_topic_score_gemma":0.003946651,"domain_scores_codex":[0.8742353,0.09865557,0.005202886,0.003412977,0.01606593,0.002427157],"domain_scores_gemma":[0.7288923,0.2184073,0.006795118,0.01140993,0.02795895,0.00653644],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001058228,0.0004708457,0.009352677,0.002286219,0.00005453258,0.0008895266,0.1256442,0.001776764,0.001659462,0.2651336,0.01500653,0.5776199],"study_design_scores_gemma":[0.00005947058,0.0005248478,0.004454812,0.007292223,0.00005254485,0.003032742,0.1477638,0.007793355,0.004473376,0.4460679,0.3782293,0.0002556165],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.0737337,0.0687464,0.3710661,0.4129272,0.004800982,0.0005438371,0.0001002796,0.0009168176,0.06716476],"genre_scores_gemma":[0.699131,0.0279727,0.2437757,0.0172046,0.002460339,0.0009033679,0.0001042226,0.000409336,0.008038647],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.1412789,"threshold_uncertainty_score":0.7471634,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1978294241","doi":"10.1108/02621710010322580","title":"Receptivity to assessment‐based feedback for management development","year":2000,"lang":"en","type":"article","venue":"Journal of Management Development","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":65,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Concordia University","funders":"","keywords":"Receptivity; Psychology; Congruence (geometry); Social psychology; Positive feedback","authors":[{"name":"Ann Marie Ryan","is_ca":false},{"name":"Stéphane Brutus","is_ca":true},{"name":"Gary J. Greguras","is_ca":false},{"name":"Milton D. Hakel","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0340642683330416,"gpt":0.3544036209878689,"spread":0.3203393526548273,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02288319,0.000337284,0.0004175846,0.0008956374,0.0005515334,0.001995642,0.0003752625,0.0005507095,0.003252852],"category_scores_gemma":[0.1268189,0.0002428482,0.0005142866,0.0002225693,0.0004098428,0.0008226979,0.001295408,0.00159783,0.0005864875],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005539984,"about_ca_system_score_gemma":0.001308431,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007902646,"about_ca_topic_score_gemma":0.0008065496,"domain_scores_codex":[0.9840013,0.01014944,0.0007152847,0.0004757989,0.004038933,0.0006192538],"domain_scores_gemma":[0.9077383,0.06371439,0.01059108,0.004346035,0.01004549,0.003564741],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0008637729,0.001571189,0.7675112,0.0003169806,0.0001371228,0.000317941,0.01908503,0.000471056,0.01066354,0.00104791,0.0009543193,0.19706],"study_design_scores_gemma":[0.00009265805,0.005279714,0.9462428,0.0006791727,0.0001651339,0.001240929,0.01608174,0.004419352,0.01546765,0.001959814,0.008217125,0.0001539405],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9912571,0.0002719446,0.002267341,0.0005808711,0.00006512586,0.0001153575,0.00005238178,0.00005417604,0.005335729],"genre_scores_gemma":[0.9975175,0.0001427172,0.001323334,0.000103433,0.00002253857,0.00006848249,0.0000324281,0.00001153631,0.0007780795],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02288319,"threshold_uncertainty_score":0.1210193,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2147628522","doi":"","title":"Some drivers of test item difficulty in mathematics : an analysis of the competency rubric","year":2012,"lang":"en","type":"article","venue":"ACEReSearch (Australian Council for Educational Research)","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":64,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Rubric; Test (biology); Inclusion (mineral); Mathematics education; Test preparation; Psychology; Pedagogy; Computer science; Medical education; Engineering; Social psychology; Medicine","authors":[{"name":"Ross Turner","is_ca":false},{"name":"Ray J Adams","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.35000712176952,"gpt":0.4785186624063204,"spread":0.1285115406368004,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03741658,0.0006228805,0.0007063803,0.008264165,0.001108666,0.004221463,0.001721224,0.0008319955,0.002523207],"category_scores_gemma":[0.1999288,0.0007479638,0.001639799,0.006322259,0.002038197,0.003215144,0.004111198,0.001890721,0.0004784578],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003058731,"about_ca_system_score_gemma":0.002692601,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01043816,"about_ca_topic_score_gemma":0.008527434,"domain_scores_codex":[0.9606615,0.0162402,0.003578163,0.002408934,0.01581968,0.00129148],"domain_scores_gemma":[0.7006871,0.207087,0.04435535,0.01366994,0.03144934,0.002751214],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00004269888,0.00005585984,0.9746601,0.00004931523,0.00009363974,0.00006636975,0.003808717,0.0004716518,0.0002852333,0.001448703,0.0003193657,0.01869841],"study_design_scores_gemma":[0.000004311712,0.0001031713,0.9897703,0.00004990664,0.00002514116,0.0001422853,0.00303944,0.004555251,0.0004035973,0.0008476386,0.001023723,0.00003529524],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9808403,0.0003375392,0.01232815,0.0005073913,0.00002297909,0.0003131847,0.0004700876,0.0000797025,0.005100605],"genre_scores_gemma":[0.9959928,0.00005905103,0.002909049,0.00002047926,0.0000101128,0.0001109973,0.0004269933,0.00003576854,0.0004347384],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03741658,"threshold_uncertainty_score":0.1978801,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2472075195","doi":"10.1017/s0261444815000233","title":"Review of washback research literature within Kane's argument-based validation framework","year":2015,"lang":"en","type":"article","venue":"Language Teaching","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":64,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Empirical research; Argument (complex analysis); Maturity (psychological); Systematic review; English language; Psychology; Political science; Epistemology; Mathematics education; Medicine; Philosophy; Law; Developmental psychology","authors":[{"name":"Liying Cheng","is_ca":true},{"name":"Youyi Sun","is_ca":true},{"name":"Jia Ma","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08897567591661655,"gpt":0.4676807480738409,"spread":0.3787050721572244,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05135575,0.001097062,0.003434882,0.01831709,0.001615299,0.007423318,0.003091957,0.003056044,0.004298945],"category_scores_gemma":[0.1683104,0.001165207,0.002071928,0.015549,0.00570466,0.01002717,0.00410487,0.003104799,0.0009539415],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005699094,"about_ca_system_score_gemma":0.01597768,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002772161,"about_ca_topic_score_gemma":0.004763206,"domain_scores_codex":[0.9562926,0.02481888,0.006562093,0.00169091,0.01000777,0.0006277218],"domain_scores_gemma":[0.7631053,0.2048813,0.009693179,0.003402306,0.0182517,0.0006661583],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.00009728548,0.0001133503,0.001236695,0.109047,0.0004675871,0.0004554709,0.009397997,0.0004234482,0.0003415936,0.02856335,0.009895733,0.8399606],"study_design_scores_gemma":[0.00008605568,0.000295413,0.007582978,0.5670429,0.001329129,0.001355223,0.01567238,0.0007069353,0.001093848,0.04126912,0.3634501,0.0001159144],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.00282772,0.9820151,0.003471089,0.004914734,0.0006040175,0.0001756233,0.0000510355,0.00002726536,0.005913349],"genre_scores_gemma":[0.0560399,0.9300365,0.009087312,0.002560428,0.0003676231,0.000533677,0.0001452979,0.00003265778,0.00119662],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9486442,"threshold_uncertainty_score":0.2715984,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2063505651","doi":"10.3200/joeb.81.6.322-326","title":"Using Peer Review to Improve Student Writing in Business Courses","year":2006,"lang":"en","type":"article","venue":"Journal of Education for Business","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":62,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Saint Mary's University","funders":"","keywords":"Peer feedback; Technical peer review; Peer review; Computer science; Peer-to-peer; Medical education; Psychology; Mathematics education; World Wide Web; Medicine; Political science","authors":[{"name":"Lloyd J. Rieber","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0662221502487459,"gpt":0.4615727496667562,"spread":0.3953505994180103,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02689581,0.000729423,0.001089317,0.002314051,0.002010687,0.003634717,0.001487597,0.001153522,0.004951826],"category_scores_gemma":[0.1721615,0.0002954288,0.0005288415,0.001359729,0.001037229,0.002465236,0.003254514,0.001917704,0.002626684],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006309659,"about_ca_system_score_gemma":0.002199726,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002695033,"about_ca_topic_score_gemma":0.0006912832,"domain_scores_codex":[0.9532444,0.0297875,0.00222512,0.000939073,0.01276989,0.001034049],"domain_scores_gemma":[0.8369567,0.1077019,0.009027643,0.007525095,0.03229142,0.006497179],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0002858286,0.00247909,0.007655326,0.0007289488,0.00004577377,0.0003576832,0.01340934,0.0005297461,0.00804319,0.001279125,0.03572593,0.92946],"study_design_scores_gemma":[0.001145734,0.01917218,0.1959324,0.004216717,0.0005181161,0.007322373,0.0692514,0.01955266,0.1081973,0.03310893,0.5405924,0.0009897631],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7951857,0.003982971,0.07782498,0.01742811,0.004847881,0.001924531,0.0001388698,0.004461546,0.09420542],"genre_scores_gemma":[0.8450234,0.003083512,0.1258445,0.001859372,0.001321658,0.0007245545,0.0001701864,0.0006024756,0.02137036],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02689581,"threshold_uncertainty_score":0.1422403,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2061239388","doi":"10.3138/cmlr.67.3.377","title":"Working Smarter, Not Working Harder: Revisiting Teacher Feedback in the L2 Writing Classroom","year":2011,"lang":"en","type":"article","venue":"Canadian Modern Language Review/ La Revue canadienne des langues vivantes","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":62,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Formative assessment; Feeling; Confusion; Psychology; Value (mathematics); Mathematics education; Peer feedback; Pedagogy; Social psychology; Computer science","authors":[{"name":"Icy Lee","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05312265572183805,"gpt":0.2858817771152325,"spread":0.2327591213933944,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03226363,0.000469903,0.000768245,0.001505684,0.002568699,0.004854318,0.001243604,0.001259296,0.0009034803],"category_scores_gemma":[0.1185957,0.0004479862,0.0002225808,0.001314999,0.003309545,0.002744817,0.002755353,0.00165339,0.0002722013],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003264796,"about_ca_system_score_gemma":0.007345907,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01559571,"about_ca_topic_score_gemma":0.02995621,"domain_scores_codex":[0.953249,0.03938608,0.001254169,0.0009989314,0.004152463,0.0009593294],"domain_scores_gemma":[0.8794365,0.09654596,0.008680123,0.003583413,0.009589012,0.002165027],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0002458236,0.000543773,0.05759037,0.0008169867,0.00003262709,0.0009876268,0.7173811,0.0002618135,0.003593537,0.001121355,0.001678088,0.2157468],"study_design_scores_gemma":[0.0001554044,0.002275167,0.1786011,0.001854477,0.0001252301,0.00107536,0.7704825,0.002201343,0.007889927,0.003503577,0.03168166,0.00015428],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9887427,0.001194839,0.002607706,0.00199517,0.00005870275,0.0001252495,0.00002120674,0.00005168271,0.005202673],"genre_scores_gemma":[0.9949896,0.0007329203,0.002726485,0.0002232598,0.00001424002,0.00007654982,0.00001435142,0.00001665819,0.001205977],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03226363,"threshold_uncertainty_score":0.1706284,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2770188235","doi":"10.1080/09585176.2017.1401550","title":"Student perspectives on assessment for learning","year":2017,"lang":"en","type":"article","venue":"The Curriculum Journal","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":62,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Mathematics education; Thematic analysis; Terminology; Psychology; Portfolio; Coding (social sciences); Value (mathematics); Computer science; Qualitative research; Mathematics; Sociology","authors":[{"name":"Christopher DeLuca","is_ca":true},{"name":"Allison E. A. Chapman-Chin","is_ca":true},{"name":"Danielle LaPointe-McEwan","is_ca":true},{"name":"Don A. Klinger","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03844282093240321,"gpt":0.4392462300933151,"spread":0.4008034091609119,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01218034,0.0003365554,0.0005444459,0.001547111,0.003004653,0.008629589,0.0007203244,0.001387316,0.003555432],"category_scores_gemma":[0.03584227,0.0002051252,0.0004671432,0.001328851,0.002011559,0.002362111,0.005131399,0.003513813,0.000924209],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002673273,"about_ca_system_score_gemma":0.002827907,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001551474,"about_ca_topic_score_gemma":0.00172886,"domain_scores_codex":[0.97412,0.01695677,0.001172266,0.0006736732,0.005173077,0.001904176],"domain_scores_gemma":[0.9607329,0.01807244,0.005305615,0.001286329,0.007953293,0.006649295],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0002564452,0.0007717198,0.1862431,0.0003375436,0.00005383186,0.0019361,0.5444784,0.0005198692,0.004335626,0.01572521,0.01379823,0.2315439],"study_design_scores_gemma":[0.00003362105,0.001112176,0.1083886,0.0007587979,0.00003819726,0.002773765,0.6224597,0.001488976,0.003636871,0.007508415,0.2516456,0.0001553138],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9407406,0.0008110243,0.002090611,0.009675036,0.0001958827,0.00005677913,0.00008772797,0.00006492375,0.04627737],"genre_scores_gemma":[0.9958329,0.000320815,0.0005174134,0.0005037736,0.00003955139,0.0000216009,0.00003051422,0.0000139621,0.002719419],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01218034,"threshold_uncertainty_score":0.06441659,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}